diff --git a/.env.example b/.env.example new file mode 100644 index 00000000..b3e3d235 --- /dev/null +++ b/.env.example @@ -0,0 +1,101 @@ +# localGPT environment configuration +# +# Copy to .env and edit. Every variable below is read by code; the value shown +# after "default:" is what the code uses when the variable is unset. +# +# cp .env.example .env +# +# For Docker use docker.env instead (it is passed with --env-file and also +# supplies build-time values for the frontend). + +# --------------------------------------------------------------------------- +# Services +# --------------------------------------------------------------------------- + +# Ollama server. Read by backend/ollama_client.py and rag_system/main.py. +# In Docker this becomes http://host.docker.internal:11434. +# default: http://localhost:11434 +OLLAMA_HOST=http://localhost:11434 + +# Base URL of the RAG API. backend/server.py builds /chat and /index from it. +# In Docker compose this is http://rag-api:8001. +# default: http://localhost:8001 +RAG_API_URL=http://localhost:8001 + +# Browser-facing URLs, read by the Next.js frontend (src/lib/api.ts). +# NEXT_PUBLIC_* values are inlined at build time, so change them before `npm run build`. +# default: http://localhost:8000 +NEXT_PUBLIC_API_URL=http://localhost:8000 +# default: http://localhost:8001 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 + +# --------------------------------------------------------------------------- +# Storage +# --------------------------------------------------------------------------- + +# SQLite database holding sessions, messages and index metadata. +# Set this to a shared path so the backend and the RAG API use one file. +# default: backend/chat_data.db (local) or /app/backend/chat_data.db (Docker) +# DB_PATH=backend/chat_data.db + +# LanceDB vector store. Defaults to the `storage.lancedb_uri` of the active +# pipeline config in rag_system/main.py. +# default: ./lancedb +# LANCEDB_PATH=./lancedb + +# --------------------------------------------------------------------------- +# Models +# --------------------------------------------------------------------------- + +# Answer generation (Ollama). Options: qwen3.6:27b (high-end, ~17GB), qwen3.5:4b (light). +# default: qwen3.5:9b +GENERATION_MODEL=qwen3.5:9b + +# Routing, triage, query decomposition, contextual enrichment and verification +# (Ollama). Light option: qwen3.5:2b. +# default: qwen3.5:4b +ENRICHMENT_MODEL=qwen3.5:4b + +# Embeddings (HuggingFace). The default is MIT-licensed, 1.2GB, 1024-dim, and +# measured best on our gold set (eval/DECISIONS.md). +# Option: Qwen/Qwen3-Embedding-4B for multilingual / long-context corpora. +# Changing this requires re-indexing every existing index - the stored vectors +# belong to the old model's vector space. localGPT records the embedding model +# on each table and refuses to query it with a different one. +# default: microsoft/harrier-oss-v1-0.6b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b + +# Reranker (HuggingFace). Only loaded when reranking is switched on - the +# default profile ships with it OFF, because the first stage above already +# outranks the cheap cross-encoder (eval/DECISIONS.md). When you do switch it +# on (UI "AI reranker" toggle, or reranker.enabled in the profile) this is the +# model that gets loaded, lazily. +# Options: BAAI/bge-reranker-v2-m3 (low latency, only pays off with a weaker +# embedder), answerdotai/answerai-colbert-small-v1, Qwen/Qwen3-Reranker-0.6B. +# default: Qwen/Qwen3-Reranker-4B +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B + +# --------------------------------------------------------------------------- +# Optional tuning +# --------------------------------------------------------------------------- + +# Seconds the backend waits for a RAG API chat response. +# default: 600 +# RAG_API_TIMEOUT=600 + +# Seconds the backend waits for a RAG API indexing run. +# default: 3600 +# RAG_API_INDEX_TIMEOUT=3600 + +# LLM backend selector for rag_system (`ollama` or `watsonx`). +# WatsonX additionally needs the WATSONX_* variables - see env.example.watsonx. +# default: ollama +# LLM_BACKEND=ollama + +# HuggingFace token, only needed for gated model downloads. +# HF_TOKEN= + +# Ollama divides its context window (and any per-request num_ctx) across its +# parallel slots โ€” with 2 slots a 32k request is served as a 16k window and +# oversized prompts are silently front-truncated. For single-user localGPT: +# OLLAMA_NUM_PARALLEL=1 diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md index 3610c914..c57e4945 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.md +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -53,7 +53,7 @@ Please include relevant error messages or logs: ## ๐Ÿ”ง Configuration - Deployment method: [Docker / Direct Python] -- Models used: [e.g. qwen3:0.6b, qwen3:8b] +- Models used: [e.g. qwen3.5:9b, qwen3.5:4b] - Document types: [e.g. PDF, DOCX, TXT] ## ๐Ÿ“Ž Additional Context diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 00000000..d17875d9 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,25 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + +jobs: + gateway-routing: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + + - name: Install test dependencies + run: pip install requests python-dotenv + + # Plain-script test (no pytest/unittest cases); exits non-zero on failure. + # backend/server.py guards its rag_system import, so only the two light + # dependencies above are needed. + - name: Run gateway routing test + run: python backend/test_gateway_routing.py diff --git a/.gitignore b/.gitignore index b3358283..05582891 100644 --- a/.gitignore +++ b/.gitignore @@ -63,8 +63,6 @@ logs/ # SQLite or other database files *.db -#backend/*.db -# backend/chat_history.db backend/chroma_db/ backend/chroma_db/** @@ -72,7 +70,21 @@ backend/chroma_db/** rag_system/documents/ *.pdf -# Ensure docker.env remains tracked +# Ensure docker.env and .env.example remain tracked !docker.env -!backend/chat_data.db +!.env.example +# Phase 0 evaluation harness (eval/) โ€” rebuildable artefacts only. +# The corpora, gold set, scripts and BASELINE.md are tracked. +eval/.eval_indexes/ +eval/results/ +!eval/corpora/*.pdf +# The Phase 4 cross-reference corpus lives one directory deeper. +!eval/corpora/acquisition/*.pdf + + +# local virtualenv +.venv/ + +# agent worktree / session state +.claude/ diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 7f281a60..469f6559 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,12 +1,13 @@ # Contributing to LocalGPT -Thank you for your interest in contributing to LocalGPT! This guide will help you get started with contributing to our private document intelligence platform. +Thank you for your interest in contributing to LocalGPT! This guide will help you get +started with contributing to our private document intelligence platform. ## ๐Ÿš€ Quick Start for Contributors ### Prerequisites -- Python 3.8+ (we test with 3.11.5) -- Node.js 16+ (we test with 23.10.0) +- Python 3.10+ (3.11 recommended) +- Node.js 20+ - Git - Ollama (for local AI models) @@ -15,44 +16,51 @@ Thank you for your interest in contributing to LocalGPT! This guide will help yo 1. **Fork and Clone** ```bash # Fork the repository on GitHub, then clone your fork - git clone https://github.com/YOUR_USERNAME/multimodal_rag.git - cd multimodal_rag - + git clone https://github.com/YOUR_USERNAME/localGPT.git + cd localGPT + # Add upstream remote - git remote add upstream https://github.com/PromtEngineer/multimodal_rag.git + git remote add upstream https://github.com/PromtEngineer/localGPT.git ``` 2. **Set Up Development Environment** ```bash # Install Python dependencies pip install -r requirements.txt - + # Install Node.js dependencies npm install - - # Install Ollama and models + + # Install Ollama and the two default models curl -fsSL https://ollama.ai/install.sh | sh - ollama pull qwen3:0.6b - ollama pull qwen3:8b + ollama pull qwen3.5:9b # generation + ollama pull qwen3.5:4b # routing / enrichment / verification ``` + Defaults work without any configuration. To override model names, service URLs or + `DB_PATH`, put them in a `.env` at the repository root โ€” `rag_system/main.py` calls + `load_dotenv()` on import, and `backend/server.py` imports it. `.env.example` lists + every variable the code actually reads. + 3. **Verify Setup** ```bash - # Run health check + # Config, agent construction, embedding model and LanceDB access python system_health_check.py - - # Start development system + + # Start Ollama + RAG API + backend + frontend python run_system.py --mode dev + + # In another shell: are all four services actually healthy? + python run_system.py --health ``` ## ๐Ÿ“‹ Development Workflow ### Branch Strategy -We use a feature branch workflow: +We use a feature branch workflow off `main`, which is the only long-lived branch: - `main` - Production-ready code -- `docker` - Docker deployment features and documentation - `feature/*` - New features - `fix/*` - Bug fixes - `docs/*` - Documentation updates @@ -64,27 +72,17 @@ We use a feature branch workflow: # Update your main branch git checkout main git pull upstream main - + # Create feature branch git checkout -b feature/your-feature-name ``` 2. **Make Your Changes** - - Follow our [coding standards](#coding-standards) - - Write tests for new functionality - - Update documentation as needed + - Follow our [coding standards](#-coding-standards) + - Update documentation as needed โ€” docs are expected to describe what the code does + today, not what it might do later -3. **Test Your Changes** - ```bash - # Run health checks - python system_health_check.py - - # Test specific components - python -m pytest tests/ -v - - # Test system integration - python run_system.py --health - ``` +3. **Check Your Changes** โ€” see [Verifying changes](#-verifying-changes) 4. **Commit Your Changes** ```bash @@ -103,25 +101,23 @@ We use a feature branch workflow: ### ๐Ÿ› Bug Fixes - Check existing issues first - Include reproduction steps -- Add tests to prevent regression +- Describe how you verified the fix ### โœจ New Features - Discuss in issues before implementing - Follow existing architecture patterns -- Include comprehensive tests - Update documentation ### ๐Ÿ“š Documentation - Fix typos and improve clarity - Add examples and use cases -- Update API documentation -- Improve setup guides +- **Verify every command, port, endpoint, config key, model name and default against the + code before writing it.** A doc that describes a feature the code does not have is worse + than no doc. ### ๐Ÿงช Testing -- Add unit tests -- Improve integration tests -- Add performance benchmarks -- Test edge cases +- There is no automated test suite yet; adding one is welcome +- Until then, describe the manual verification you ran in the PR ## ๐Ÿ“ Coding Standards @@ -131,30 +127,38 @@ We follow PEP 8 with some modifications: ```python # Use type hints -def process_document(file_path: str, config: Dict[str, Any]) -> ProcessingResult: - """Process a document with the given configuration. - +def convert_to_markdown(self, file_path: str) -> List[Tuple[str, Dict[str, Any]]]: + """Convert a document to Markdown, preserving layout and tables. + Args: file_path: Path to the document file - config: Processing configuration dictionary - + Returns: - ProcessingResult object with metadata and chunks + A list of (markdown, metadata) tuples, optionally with the + DoclingDocument as a third element. """ - pass + ... # Use descriptive variable names -embedding_model_name = "Qwen/Qwen3-Embedding-0.6B" -retrieval_results = retriever.search(query, top_k=20) - -# Use dataclasses for structured data -@dataclass -class IndexingConfig: - embedding_batch_size: int = 50 - enable_late_chunking: bool = True - chunk_size: int = 512 +embedding_model_name = "microsoft/harrier-oss-v1-0.6b" +retrieved_docs = retriever.retrieve(text_query=query, table_name=table, k=20) ``` +Conventions specific to this codebase: + +- **Configuration is plain dicts**, defined once in `rag_system/main.py` and handed out as + deep copies by `rag_system/factory.py::get_pipeline_config()`. Do not introduce a second + place where a model name or a default lives. +- **Every config key must be read by code.** If you add a key, wire it; if you find a key + nothing reads, delete it rather than documenting it. +- **No hardcoded model names or embedding dimensions.** Resolve models from + `OLLAMA_CONFIG` / `EXTERNAL_MODELS` (which are environment-overridable), and derive + vector width from the embeddings the loaded model produced. +- **Fail loudly on misconfiguration, degrade quietly on optional components.** A missing + `embedding_model_name` raises; a reranker that cannot be loaded logs a warning and the + query proceeds without reranking. +- Keep comments purposeful. No decorative banners for trivial code. + ### TypeScript/React Code Style ```typescript @@ -162,19 +166,18 @@ class IndexingConfig: interface ChatMessage { id: string; content: string; - role: 'user' | 'assistant'; - timestamp: Date; - sources?: DocumentSource[]; + sender: 'user' | 'assistant'; + timestamp: string; } // Use functional components with hooks const ChatInterface: React.FC = ({ sessionId }) => { const [messages, setMessages] = useState([]); - + const handleSendMessage = useCallback(async (content: string) => { // Implementation }, [sessionId]); - + return (
{/* Component JSX */} @@ -183,126 +186,122 @@ const ChatInterface: React.FC = ({ sessionId }) => { }; ``` +- All API calls go through `src/lib/api.ts`. Base URLs come from `NEXT_PUBLIC_API_URL` and + `NEXT_PUBLIC_RAG_API_URL`; never hardcode a host. +- Response types in `src/lib/api.ts` must match what the server actually returns. + ### File Organization ``` rag_system/ -โ”œโ”€โ”€ agent/ # ReAct agent implementation -โ”œโ”€โ”€ indexing/ # Document processing and indexing -โ”œโ”€โ”€ retrieval/ # Search and retrieval components -โ”œโ”€โ”€ pipelines/ # End-to-end processing pipelines -โ”œโ”€โ”€ rerankers/ # Result reranking implementations -โ””โ”€โ”€ utils/ # Shared utilities +โ”œโ”€โ”€ main.py # Master configuration + CLI +โ”œโ”€โ”€ factory.py # The single factory (get_agent / get_indexing_pipeline) +โ”œโ”€โ”€ api_server.py # HTTP API on :8001 +โ”œโ”€โ”€ agent/ # Triage, decomposition, orchestration and verification loop +โ”œโ”€โ”€ ingestion/ # Document conversion (Docling) and chunking +โ”œโ”€โ”€ indexing/ # Embedding, LanceDB writing, enrichment, overviews +โ”œโ”€โ”€ retrieval/ # Retrievers and query transformation +โ”œโ”€โ”€ pipelines/ # End-to-end indexing and retrieval pipelines +โ”œโ”€โ”€ rerankers/ # Reranking and sentence pruning +โ””โ”€โ”€ utils/ # LLM clients and shared helpers + +backend/ # Gateway on :8000 (sessions, uploads, SQLite) src/ โ”œโ”€โ”€ components/ # React components -โ”œโ”€โ”€ lib/ # Utility functions and API clients -โ””โ”€โ”€ app/ # Next.js app router pages +โ”œโ”€โ”€ lib/ # API client and shared types +โ”œโ”€โ”€ utils/ # Small helpers +โ””โ”€โ”€ app/ # Next.js app router pages ``` -## ๐Ÿงช Testing Guidelines +## ๐Ÿงช Verifying changes -### Unit Tests -```python -# Test file: tests/test_embeddings.py -import pytest -from rag_system.indexing.embedders import HuggingFaceEmbedder - -def test_embedding_generation(): - embedder = HuggingFaceEmbedder("sentence-transformers/all-MiniLM-L6-v2") - embeddings = embedder.create_embeddings(["test text"]) - - assert embeddings.shape[0] == 1 - assert embeddings.shape[1] == 384 # Model dimension - assert embeddings.dtype == np.float32 +There is no `tests/` directory and no pytest suite. Use these checks: + +### Python +```bash +# Syntax check everything +find rag_system backend -name '*.py' -exec python -m py_compile {} + + +# Config, agent construction, embedding model, LanceDB access, sample query +python system_health_check.py + +# Are the running services healthy? (exits non-zero if not) +python run_system.py --health + +# Exercise a pipeline directly +python -m rag_system.main index ./path/to/docs --mode fast +python -m rag_system.main chat "test question" --mode fast ``` -### Integration Tests -```python -# Test file: tests/test_integration.py -def test_end_to_end_indexing(): - """Test complete document indexing pipeline.""" - agent = get_agent("test") - result = agent.index_documents(["test_document.pdf"]) - - assert result.success - assert len(result.indexed_chunks) > 0 +### Frontend +```bash +npx tsc --noEmit # type check +npm run lint # Next.js ESLint +npm run build # production build ``` -### Frontend Tests -```typescript -// Test file: src/components/__tests__/ChatInterface.test.tsx -import { render, screen, fireEvent } from '@testing-library/react'; -import { ChatInterface } from '../ChatInterface'; - -test('sends message when form is submitted', async () => { - render(); - - const input = screen.getByPlaceholderText('Type your message...'); - const button = screen.getByRole('button', { name: /send/i }); - - fireEvent.change(input, { target: { value: 'test message' } }); - fireEvent.click(button); - - expect(screen.getByText('test message')).toBeInTheDocument(); -}); +`next.config.ts` sets `eslint.ignoreDuringBuilds` and `typescript.ignoreBuildErrors`, so a +successful `npm run build` does **not** imply the code type-checks. Run `npx tsc --noEmit` +separately. + +### Docker +```bash +./test_docker_build.sh +``` + +### End-to-end smoke test +```bash +curl http://localhost:8000/health +curl http://localhost:8001/health +curl -X POST http://localhost:8001/chat \ + -H 'Content-Type: application/json' \ + -d '{"query": "what is this document about?"}' ``` ## ๐Ÿ“– Documentation Standards ### Code Documentation ```python -def create_index( - documents: List[str], - config: IndexingConfig, - progress_callback: Optional[Callable[[float], None]] = None -) -> IndexingResult: - """Create a searchable index from documents. - - This function processes documents through the complete indexing pipeline: - 1. Text extraction and chunking - 2. Embedding generation - 3. Vector database storage - 4. BM25 index creation - +def run(self, file_paths: List[str] | None = None, *, documents: List[str] | None = None): + """Process and index documents according to the pipeline configuration. + + Steps: Docling conversion -> chunking -> optional contextual enrichment -> + embedding into LanceDB (plus the native FTS index) -> optional late chunking. + Args: - documents: List of document file paths to index - config: Indexing configuration with model settings and parameters - progress_callback: Optional callback function for progress updates - - Returns: - IndexingResult containing success status, metrics, and any errors - + file_paths: Absolute paths of the documents to index + documents: Legacy alias for file_paths + Raises: - IndexingError: If document processing fails - ModelLoadError: If embedding model cannot be loaded - - Example: - >>> config = IndexingConfig(embedding_batch_size=32) - >>> result = create_index(["doc1.pdf", "doc2.pdf"], config) - >>> print(f"Indexed {result.chunk_count} chunks") + TypeError: If neither argument is supplied + ValueError: If the embeddings do not match the target table's vector width """ ``` -### API Documentation +### HTTP handler documentation + +Both servers are built on the standard library's `http.server`; there is no FastAPI, no +Pydantic models and no generated OpenAPI schema. Document a route with a handler docstring +and keep the field list in the relevant README in sync: + ```python -# Use OpenAPI/FastAPI documentation -@app.post("/chat", response_model=ChatResponse) -async def chat_endpoint(request: ChatRequest) -> ChatResponse: - """Chat with indexed documents. - - Send a natural language query and receive an AI-generated response - based on the indexed document collection. - - - **query**: The user's question or prompt - - **session_id**: Chat session identifier - - **search_type**: Type of search (vector, hybrid, bm25) - - **retrieval_k**: Number of documents to retrieve - - Returns a response with the AI-generated answer and source documents. +def handle_chat(self): + """POST /chat โ€” answer a query with the agentic RAG pipeline. + + Body (camelCase accepted, normalised to snake_case): + query (required), session_id, table_name, model, + retrieval_mode (hybrid|vector_only|fts_only), force_rag, + query_decompose, ai_rerank, context_expand, verify, + retrieval_k, context_window_size, reranker_top_k. + + Returns {"answer": str, "source_documents": list}. """ ``` +Route tables live in [`backend/README.md`](backend/README.md) (port 8000) and +[`rag_system/DOCUMENTATION.md`](rag_system/DOCUMENTATION.md) (port 8001). + ## ๐Ÿ”ง Development Tools ### Recommended VS Code Extensions @@ -310,44 +309,20 @@ async def chat_endpoint(request: ChatRequest) -> ChatResponse: { "recommendations": [ "ms-python.python", - "ms-python.pylint", - "ms-python.black-formatter", "bradlc.vscode-tailwindcss", - "esbenp.prettier-vscode", "ms-vscode.vscode-typescript-next" ] } ``` -### Pre-commit Hooks -```bash -# Install pre-commit -pip install pre-commit - -# Set up hooks -pre-commit install - -# Run manually -pre-commit run --all-files -``` - -### Development Scripts -```bash -# Lint Python code -python -m pylint rag_system/ - -# Format Python code -python -m black rag_system/ - -# Type check -python -m mypy rag_system/ +### Formatters and linters -# Lint TypeScript -npm run lint +The repository ships an ESLint flat config (`eslint.config.mjs`, used by `npm run lint`) +and nothing else โ€” there is no Black, pylint, mypy, Prettier or pre-commit configuration +checked in, and `npm run format` does not exist. If you run a Python formatter locally, +keep the diff limited to the lines you actually changed. -# Format TypeScript -npm run format -``` +Available npm scripts (`package.json`): `dev`, `build`, `start`, `lint`. ## ๐Ÿ› Issue Reporting @@ -356,10 +331,11 @@ When reporting bugs, please include: 1. **Environment Information** ``` - - OS: macOS 13.4 + - OS: macOS 15.5 - Python: 3.11.5 - - Node.js: 23.10.0 + - Node.js: 20.11.0 - Ollama: 0.9.5 + - LLM_BACKEND: ollama ``` 2. **Steps to Reproduce** @@ -371,7 +347,8 @@ When reporting bugs, please include: ``` 3. **Expected vs Actual Behavior** -4. **Error Messages and Logs** +4. **Error Messages and Logs** โ€” the RAG API prints most of its pipeline trace to stdout; + `RAG_LOG_LEVEL=DEBUG` adds more 5. **Screenshots (if applicable)** ### Feature Requests @@ -391,11 +368,12 @@ We use semantic versioning (semver): - Patch: Bug fixes ### Release Checklist -- [ ] All tests pass -- [ ] Documentation updated +- [ ] `system_health_check.py` and `run_system.py --health` pass +- [ ] `npx tsc --noEmit` and `npm run build` pass +- [ ] Documentation updated and re-verified against the code - [ ] Version bumped in relevant files - [ ] Changelog updated -- [ ] Docker images built and tested +- [ ] Docker images built and tested (`./test_docker_build.sh`) - [ ] Release notes prepared ## ๐Ÿค Community Guidelines @@ -418,8 +396,8 @@ We use semantic versioning (semver): 1. **Performance Optimization**: Improving indexing and retrieval speed 2. **Model Support**: Adding more embedding and generation models 3. **User Experience**: Enhancing the web interface -4. **Documentation**: Improving setup and usage guides -5. **Testing**: Expanding test coverage +4. **Documentation**: Keeping setup and usage guides truthful +5. **Testing**: Establishing an automated test suite ### Architecture Goals - **Modularity**: Components should be loosely coupled @@ -430,23 +408,29 @@ We use semantic versioning (semver): ## ๐Ÿ“š Additional Resources -### Learning Resources -- [RAG System Architecture Overview](Documentation/architecture_overview.md) +### Project documentation +- [RAG package overview](rag_system/README.md) +- [RAG reference: pipelines, config keys, HTTP API](rag_system/DOCUMENTATION.md) +- [Backend gateway](backend/README.md) +- [Watson X backend](WATSONX_README.md) +- [Architecture Overview](Documentation/architecture_overview.md) - [API Reference](Documentation/api_reference.md) - [Deployment Guide](Documentation/deployment_guide.md) -- [Troubleshooting Guide](DOCKER_TROUBLESHOOTING.md) +- [Docker Troubleshooting](DOCKER_TROUBLESHOOTING.md) ### External References -- [LangChain Documentation](https://python.langchain.com/) -- [Ollama Documentation](https://ollama.ai/docs) +- [Ollama](https://github.com/ollama/ollama) +- [Docling](https://github.com/docling-project/docling) +- [LanceDB](https://github.com/lancedb/lancedb) +- [rerankers](https://github.com/AnswerDotAI/rerankers) - [Next.js Documentation](https://nextjs.org/docs) -- [FastAPI Documentation](https://fastapi.tiangolo.com/) --- ## ๐Ÿ™ Thank You! -Thank you for contributing to LocalGPT! Your contributions help make private document intelligence accessible to everyone. +Thank you for contributing to LocalGPT! Your contributions help make private document +intelligence accessible to everyone. For questions about contributing, please: 1. Check existing documentation @@ -454,4 +438,4 @@ For questions about contributing, please: 3. Create a new issue with the `question` label 4. Join our community discussions -Happy coding! ๐Ÿš€ \ No newline at end of file +Happy coding! ๐Ÿš€ diff --git a/DOCKER_README.md b/DOCKER_README.md index bc78e8b5..fb2f652e 100644 --- a/DOCKER_README.md +++ b/DOCKER_README.md @@ -1,10 +1,11 @@ # ๐Ÿณ LocalGPT Docker Deployment Guide -This guide covers running LocalGPT using Docker containers with local Ollama for optimal performance. +This guide covers running LocalGPT in Docker containers, with Ollama either on the +host (default, best performance) or as a container. ## ๐Ÿš€ Quick Start -### Complete Setup (5 Minutes) +### Complete Setup ```bash # 1. Install Ollama locally curl -fsSL https://ollama.ai/install.sh | sh @@ -12,121 +13,215 @@ curl -fsSL https://ollama.ai/install.sh | sh # 2. Start Ollama server ollama serve -# 3. Install required models (in another terminal) -ollama pull qwen3:0.6b -ollama pull qwen3:8b +# 3. Install the models (in another terminal) +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, enrichment, verification # 4. Clone and start LocalGPT -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT ./start-docker.sh # 5. Access the application open http://localhost:3000 ``` +The first `up --build` installs the Python and Node dependencies and then the +`rag-api` container downloads its embedding model from HuggingFace (the reranker follows lazily on the first reranked query), +so allow several minutes before the UI is reachable. + +Running this from a script or CI? Add `-y` so the fallback prompt never blocks: + +```bash +./start-docker.sh local -y # or: NONINTERACTIVE=1 ./start-docker.sh +``` + ## ๐Ÿ“‹ Prerequisites -- **Docker Desktop** installed and running -- **Ollama** installed locally (required for best performance) -- **8GB+ RAM** (16GB recommended for larger models) -- **10GB+ free disk space** +- **Docker Desktop** (or Docker Engine 24+ with the Compose plugin), running +- **Ollama** on the host (recommended) or the containerized fallback below +- **8GB+ RAM** (16GB recommended) +- **20GB+ free disk space** โ€” images, plus ~10GB of HuggingFace weights inside + `rag-api`, plus the Ollama models ## ๐Ÿ—๏ธ Architecture -### Current Setup (Local Ollama + Docker Containers) +### Default Setup (Local Ollama + Docker Containers) ``` โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ” โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ” โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ” โ”‚ Frontend โ”‚โ”€โ”€โ”€โ”€โ”‚ Backend โ”‚โ”€โ”€โ”€โ”€โ”‚ RAG API โ”‚ โ”‚ (Container) โ”‚ โ”‚ (Container) โ”‚ โ”‚ (Container) โ”‚ โ”‚ Port: 3000 โ”‚ โ”‚ Port: 8000 โ”‚ โ”‚ Port: 8001 โ”‚ โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ - โ”‚ - โ”‚ API calls - โ–ผ - โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ” - โ”‚ Ollama โ”‚ - โ”‚ (Local/Host) โ”‚ - โ”‚ Port: 11434 โ”‚ - โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ + โ”‚ โ”‚ + โ”‚ browser streams directly to :8001 โ”‚ + โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”ค + โ”‚ API calls + โ–ผ + โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ” + โ”‚ Ollama โ”‚ + โ”‚ (Host, default) โ”‚ + โ”‚ Port: 11434 โ”‚ + โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ ``` -**Why Local Ollama?** +**Why host Ollama by default?** - โœ… Better performance (direct GPU access) - โœ… Simpler setup (one less container) - โœ… Easier model management -- โœ… More reliable connection +- โœ… Models survive `docker system prune` + +### Containerized Ollama + +`docker-compose.yml` also defines an `ollama` service behind the `with-ollama` +profile, for machines where you would rather not install Ollama: + +```bash +./start-docker.sh container +# equivalently: +# OLLAMA_HOST=http://ollama:11434 \ +# docker compose --env-file docker.env --profile with-ollama up --build -d + +# Pull the models inside the container the first time +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` + +Its models persist in the named volume `ollama_data`. + +### Startup Order + +``` +rag-api (healthy: GET /health) โ†’ backend (healthy: GET /health) โ†’ frontend +``` + +`backend` declares `depends_on: rag-api: condition: service_healthy`, and `rag-api` +does not answer `/health` until the agent and the embedding model are loaded (the +reranker is fetched lazily on the first query that reranks). +Seeing `backend` sit in `created` for a few minutes on a cold start is normal โ€” +follow `docker compose logs -f rag-api`. ## ๐Ÿ› ๏ธ Container Details ### Frontend Container (rag-frontend) -- **Image**: Custom Node.js 18 build +- **Image**: `node:20-alpine`, built by `Dockerfile.frontend` - **Port**: 3000 - **Purpose**: Next.js web interface -- **Health Check**: HTTP GET to / -- **Memory**: ~500MB +- **Health Check**: busybox `wget -qO- http://localhost:3000` (the alpine image has no curl) +- **Build args**: `NEXT_PUBLIC_API_URL`, `NEXT_PUBLIC_RAG_API_URL` โ€” inlined by `next build` -### Backend Container (rag-backend) -- **Image**: Custom Python 3.11 build +### Backend Container (rag-backend) +- **Image**: `python:3.11-slim`, built by `Dockerfile.backend` - **Port**: 8000 -- **Purpose**: Session management, chat history, API gateway -- **Health Check**: HTTP GET to /health -- **Memory**: ~300MB +- **Purpose**: sessions, indexes, uploads, chat history, gateway to the RAG API +- **Health Check**: `curl -f http://localhost:8000/health` +- **Working directory**: `/app`, started with `python backend/server.py` +- **Extras**: `sqlite3` CLI installed for database troubleshooting ### RAG API Container (rag-api) -- **Image**: Custom Python 3.11 build +- **Image**: `python:3.11-slim`, built by `Dockerfile.rag-api` - **Port**: 8001 -- **Purpose**: Document indexing, retrieval, AI processing -- **Health Check**: HTTP GET to /models -- **Memory**: ~2GB (varies with model usage) +- **Purpose**: document indexing, retrieval, the agent loop +- **Health Check**: `curl -f http://localhost:8001/health` +- **Memory**: dominated by the resident embedding and reranker models +- **Concurrency**: single-threaded โ€” one chat or indexing run at a time ## ๐Ÿ“‚ Volume Mounts & Data -### Persistent Data -- `./lancedb/` โ†’ Vector database storage -- `./index_store/` โ†’ Document indexes and metadata -- `./shared_uploads/` โ†’ Uploaded document files -- `./backend/chat_data.db` โ†’ SQLite chat history database +### Persistent Data (bind mounts to host directories) +- `./lancedb/` โ†’ `/app/lancedb` โ€” vectors and the native full-text index +- `./index_store/` โ†’ `/app/index_store` โ€” document overviews +- `./shared_uploads/` โ†’ `/app/shared_uploads` โ€” uploaded documents +- `./backend/` โ†’ `/app/backend` โ€” `chat_data.db` + +### Named volumes +- `ollama_data` โ€” only used by the optional containerized Ollama ### Shared Between Containers -All containers share access to document storage and databases through bind mounts. +`backend` and `rag-api` mount the same four directories and both set +`DB_PATH=/app/backend/chat_data.db`, so they share one SQLite file and one vector +store. HuggingFace model weights are **not** mounted โ€” they live in the container's +writable layer and are re-downloaded whenever `rag-api` is recreated. ## ๐Ÿ”ง Configuration ### Environment Variables (docker.env) ```bash -# Ollama Configuration +# Ollama on the host. Every service declares +# extra_hosts: ["host.docker.internal:host-gateway"], so this resolves on Linux too. OLLAMA_HOST=http://host.docker.internal:11434 +# Containerized alternative: OLLAMA_HOST=http://ollama:11434 -# Service Configuration +# Service wiring NODE_ENV=production RAG_API_URL=http://rag-api:8001 + +# Browser-facing URLs, inlined into the frontend at build time NEXT_PUBLIC_API_URL=http://localhost:8000 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 -# Database Paths (inside containers) -DATABASE_PATH=/app/backend/chat_data.db +# Shared SQLite database and vector store +DB_PATH=/app/backend/chat_data.db LANCEDB_PATH=/app/lancedb -UPLOADS_PATH=/app/shared_uploads + +# Models +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B +``` + +Pass it explicitly (`start-docker.sh` does this for you): + +```bash +docker compose --env-file docker.env up --build -d +``` + +Variables already exported in your shell take precedence over `--env-file`. + +`NEXT_PUBLIC_*` are **build-time** for Next.js. `docker-compose.yml` forwards them +to `Dockerfile.frontend` as build args; changing them at runtime does nothing to an +already-built image, so rebuild: + +```bash +NEXT_PUBLIC_API_URL=https://gpt.example.com/api \ +NEXT_PUBLIC_RAG_API_URL=https://gpt.example.com/rag \ +docker compose --env-file docker.env up -d --build frontend ``` ### Model Configuration -The system uses these models by default: -- **Embedding**: `Qwen/Qwen3-Embedding-0.6B` (1024 dimensions) -- **Generation**: `qwen3:0.6b` (fast) or `qwen3:8b` (high quality) -- **Reranking**: Built-in cross-encoder + +| Role | Default | Documented options | +|------|---------|--------------------| +| Generation | `qwen3.5:9b` (Ollama) | `qwen3.6:27b` (high-end, ~17GB), `qwen3.5:4b` (light) | +| Enrichment / utility | `qwen3.5:4b` (Ollama) | `qwen3.5:2b` (light) | +| Embedding | `microsoft/harrier-oss-v1-0.6b` โ€” HuggingFace, MIT, 1024 dims | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context), `Qwen/Qwen3-Embedding-0.6B` (1024 dims) | +| Reranking (**on by default**) | `Qwen/Qwen3-Reranker-4B` โ€” HuggingFace, loaded lazily on the first reranked query | `BAAI/bge-reranker-v2-m3`, `answerdotai/answerai-colbert-small-v1`, `Qwen/Qwen3-Reranker-0.6B` | + +Only the two Ollama models need `ollama pull`. Embedding dimensions are read from +the loaded model, never hardcoded โ€” **changing `EMBEDDING_MODEL` requires rebuilding +existing indexes**, and appending mismatched vectors to a LanceDB table raises an +explicit error. If the reranker cannot be loaded, the pipeline logs a warning and +continues without reranking. ## ๐ŸŽฏ Management Commands ### Start/Stop Services ```bash -# Start all services +# Start all services (local Ollama) ./start-docker.sh +# Start with containerized Ollama +./start-docker.sh container + # Stop all services ./start-docker.sh stop # Restart services ./start-docker.sh stop && ./start-docker.sh + +# Non-interactive (CI) +./start-docker.sh local -y ``` ### Monitor Services @@ -153,20 +248,23 @@ docker compose --env-file docker.env up --build -d # Stop manually docker compose down -# Rebuild specific service +# Rebuild a specific service (code is COPY-ed in, not mounted - +# a restart alone will not pick up source changes) docker compose build --no-cache rag-api docker compose up -d rag-api ``` ### Health Checks ```bash -# Test all endpoints curl -f http://localhost:3000 && echo "โœ… Frontend OK" curl -f http://localhost:8000/health && echo "โœ… Backend OK" -curl -f http://localhost:8001/models && echo "โœ… RAG API OK" +curl -f http://localhost:8001/health && echo "โœ… RAG API OK" curl -f http://localhost:11434/api/tags && echo "โœ… Ollama OK" ``` +`test_docker_build.sh` builds and smoke-tests each image individually against the +same endpoints. + ## ๐Ÿž Debugging ### Access Container Shells @@ -177,7 +275,7 @@ docker compose exec rag-api bash # Backend container docker compose exec backend bash -# Frontend container +# Frontend container (alpine -> sh, not bash) docker compose exec frontend sh ``` @@ -185,31 +283,32 @@ docker compose exec frontend sh ```bash # Test RAG system initialization docker compose exec rag-api python -c " -from rag_system.main import get_agent +from rag_system.factory import get_agent agent = get_agent('default') print('โœ… RAG System OK') " -# Test Ollama connection from container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +# Test Ollama connection from the container +docker compose exec rag-api curl -s http://host.docker.internal:11434/api/tags -# Check environment variables -docker compose exec rag-api env | grep OLLAMA +# Check the wiring the containers actually got +docker compose exec rag-api env | grep -E "OLLAMA|MODEL|DB_PATH|LANCEDB" +docker compose exec backend env | grep RAG_API_URL + +# Verify the backend can reach the RAG API by service name +docker compose exec backend curl -s http://rag-api:8001/health # View Python packages -docker compose exec rag-api pip list | grep -E "(torch|transformers|lancedb)" +docker compose exec rag-api pip list | grep -E "(torch|transformers|lancedb|docling|rerankers)" ``` ### Resource Monitoring ```bash -# Monitor container resources docker stats -# Check disk usage docker system df -df -h ./lancedb ./shared_uploads +du -sh lancedb shared_uploads index_store -# Check memory usage by service docker stats --format "table {{.Name}}\t{{.CPUPerc}}\t{{.MemUsage}}\t{{.MemPerc}}" ``` @@ -219,7 +318,7 @@ docker stats --format "table {{.Name}}\t{{.CPUPerc}}\t{{.MemUsage}}\t{{.MemPerc} #### Container Won't Start ```bash -# Check logs for specific error +# Check logs for the specific error docker compose logs [service-name] # Rebuild from scratch @@ -231,42 +330,60 @@ docker system prune -f lsof -i :3000 -i :8000 -i :8001 ``` +#### `backend` never starts +It is waiting for `rag-api` to report healthy. Check: +```bash +docker compose ps +docker compose logs -f rag-api +docker compose exec rag-api curl -s http://localhost:8001/health +``` + #### Can't Connect to Ollama ```bash -# Verify Ollama is running +# Verify Ollama is running on the host curl http://localhost:11434/api/tags # Restart Ollama pkill ollama ollama serve -# Test from container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +# Test from the container +docker compose exec rag-api curl -s http://host.docker.internal:11434/api/tags +``` + +#### Every chat answers "Could not connect to the RAG API server" +The backend builds its URLs from `RAG_API_URL`; inside compose that must be +`http://rag-api:8001`, since `localhost` there is the backend container itself. +```bash +docker compose exec backend env | grep RAG_API_URL +docker compose exec backend curl -s http://rag-api:8001/health ``` #### Memory Issues ```bash -# Check memory usage docker stats --no-stream -free -h # On host +free -h # on the host -# Increase Docker memory limit # Docker Desktop โ†’ Settings โ†’ Resources โ†’ Memory โ†’ 8GB+ -# Use smaller models -ollama pull qwen3:0.6b # Instead of qwen3:8b +# Lighter configuration +GENERATION_MODEL=qwen3.5:4b \ +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ +docker compose --env-file docker.env up -d rag-api backend +# (changing the embedding model requires rebuilding your indexes) ``` #### Frontend Build Errors ```bash -# Clean build docker compose build --no-cache frontend docker compose up -d frontend - -# Check frontend logs docker compose logs frontend ``` +#### Frontend talks to the wrong host +`NEXT_PUBLIC_API_URL` / `NEXT_PUBLIC_RAG_API_URL` are baked in at build time. +Rebuild the frontend image after changing them. + #### Database/Storage Issues ```bash # Check file permissions @@ -277,64 +394,60 @@ ls -la lancedb/ chmod 664 backend/chat_data.db chmod -R 755 lancedb/ shared_uploads/ -# Test database access +# Inspect the database (sqlite3 is installed in the image) docker compose exec backend sqlite3 /app/backend/chat_data.db ".tables" ``` -### Performance Issues - -#### Slow Response Times -- Use faster models: `qwen3:0.6b` instead of `qwen3:8b` -- Increase Docker memory allocation -- Ensure SSD storage for databases -- Monitor with `docker stats` +### Performance Notes -#### High Memory Usage -- Reduce batch sizes in configuration -- Use smaller embedding models -- Clear unused Docker resources: `docker system prune` +- The RAG API is single-threaded: a chat request queued behind an indexing run + looks like a hang. Check `docker compose logs -f rag-api`. +- Contextual enrichment makes one LLM call per chunk; it dominates indexing time. + Turn it off in the index build options for the fastest ingest. +- `RAG_CONFIG_MODE=fast` switches the RAG API to vector-only retrieval with no + reranking, decomposition or verification. ### Complete Reset ```bash -# Nuclear option - reset everything +# Nuclear option - resets everything, including your documents and chat history ./start-docker.sh stop +docker compose down +rm -rf lancedb/* index_store/* shared_uploads/* backend/chat_data.db docker system prune -a --volumes -rm -rf lancedb/* shared_uploads/* backend/chat_data.db ./start-docker.sh ``` +`docker compose down -v` alone only drops the `ollama_data` volume โ€” application +data is in host directories and must be deleted explicitly. + ## ๐Ÿ† Success Criteria Your Docker deployment is successful when: -- โœ… `./start-docker.sh status` shows all containers healthy -- โœ… All health checks pass (see commands above) +- โœ… `docker compose ps` shows all services healthy +- โœ… All health checks pass (see commands above) - โœ… You can access http://localhost:3000 - โœ… You can upload documents and create indexes - โœ… You can chat with your documents - โœ… No errors in container logs -### Performance Benchmarks - -**Good Performance:** -- Container startup: < 2 minutes -- Index creation: < 2 min per 100MB document -- Query response: < 30 seconds -- Memory usage: < 4GB total containers +### What to Expect -**Optimal Performance:** -- Container startup: < 1 minute -- Index creation: < 1 min per 100MB document -- Query response: < 10 seconds -- Memory usage: < 2GB total containers +- **First build**: installs torch/transformers/docling and builds the Next.js + bundle; then `rag-api` downloads the ~1.2GB embedding model before it reports + healthy (the ~7.5GB reranker downloads lazily, on the first reranked query) +- **Restarts**: fast. **Recreating** `rag-api` re-downloads the model weights, + because no volume is mounted for the HuggingFace cache +- **Concurrency**: one RAG request at a time ## ๐Ÿ“š Additional Resources - **Detailed Troubleshooting**: See `DOCKER_TROUBLESHOOTING.md` -- **Complete Documentation**: See `Documentation/docker_usage.md` +- **Complete Docker Guide**: See `Documentation/docker_usage.md` +- **Deployment Guide**: See `Documentation/deployment_guide.md` - **System Architecture**: See `Documentation/architecture_overview.md` -- **Direct Development**: See main `README.md` for non-Docker setup +- **Direct Development**: See the main `README.md` --- -**Happy Dockerizing! ๐Ÿณ** Need help? Check the troubleshooting guide or open an issue. \ No newline at end of file +**Happy Dockerizing! ๐Ÿณ** Need help? Check the troubleshooting guide or open an issue. diff --git a/DOCKER_TROUBLESHOOTING.md b/DOCKER_TROUBLESHOOTING.md index 4d758db5..55f49458 100644 --- a/DOCKER_TROUBLESHOOTING.md +++ b/DOCKER_TROUBLESHOOTING.md @@ -1,8 +1,9 @@ # ๐Ÿณ Docker Troubleshooting Guide - LocalGPT -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ -This guide helps diagnose and fix Docker-related issues with LocalGPT's containerized deployment. +This guide helps diagnose and fix Docker-related issues with LocalGPT's +containerized deployment. --- @@ -13,7 +14,7 @@ This guide helps diagnose and fix Docker-related issues with LocalGPT's containe # Check Docker daemon docker version -# Check Ollama status +# Check Ollama status curl http://localhost:11434/api/tags # Check containers @@ -22,7 +23,7 @@ curl http://localhost:11434/api/tags # Test all endpoints curl -f http://localhost:3000 && echo "โœ… Frontend OK" curl -f http://localhost:8000/health && echo "โœ… Backend OK" -curl -f http://localhost:8001/models && echo "โœ… RAG API OK" +curl -f http://localhost:8001/health && echo "โœ… RAG API OK" curl -f http://localhost:11434/api/tags && echo "โœ… Ollama OK" ``` @@ -34,6 +35,19 @@ curl -f http://localhost:11434/api/tags && echo "โœ… Ollama OK" โœ… Ollama OK ``` +### Know the startup order before you debug + +``` +rag-api (healthy: GET /health) โ†’ backend (healthy: GET /health) โ†’ frontend +``` + +`backend` has `depends_on: rag-api: condition: service_healthy`, and `rag-api` only +answers `/health` once the agent and the embedding model have loaded (the reranker +loads lazily, on the first reranked query). +On a cold start that is a multi-minute HuggingFace download. **`backend` sitting in +`created` is usually patience, not a bug** โ€” confirm with +`docker compose logs -f rag-api`. + --- ## ๐Ÿšจ Common Issues & Solutions @@ -64,16 +78,9 @@ docker version #### Solution B: Linux Docker Service ```bash -# Check Docker service status sudo systemctl status docker - -# Restart Docker service sudo systemctl restart docker - -# Enable auto-start sudo systemctl enable docker - -# Test connection docker version ``` @@ -84,7 +91,7 @@ sudo pkill -f docker # Remove socket files sudo rm -f /var/run/docker.sock -sudo rm -f /Users/prompt/.docker/run/docker.sock # macOS +sudo rm -f "$HOME/.docker/run/docker.sock" # macOS # Restart Docker Desktop open -a Docker # macOS @@ -99,37 +106,93 @@ ConnectionError: Failed to connect to Ollama at http://host.docker.internal:1143 #### Solution A: Verify Ollama is Running ```bash -# Check if Ollama is running curl http://localhost:11434/api/tags # If not running, start it ollama serve -# Install required models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +# Install the required models +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` +`run_system.py` pulls these automatically, but `start-docker.sh` does not โ€” with +host Ollama you must pull them yourself. + #### Solution B: Test from Container ```bash -# Test Ollama connection from RAG API container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +docker compose exec rag-api curl -sv http://host.docker.internal:11434/api/tags -# If this fails, check Docker network settings +# Inspect the network. The compose project name is the directory name, +# so the network is localgpt_rag-network. docker network ls -docker network inspect rag_system_old_default +docker network inspect localgpt_rag-network ``` -#### Solution C: Alternative Ollama Host +All services declare `extra_hosts: ["host.docker.internal:host-gateway"]`, so this +name resolves on Linux as well as macOS and Windows. If it still fails, check that +Ollama is listening on all interfaces rather than only the loopback: + ```bash -# Edit docker.env to use different host -echo "OLLAMA_HOST=http://172.17.0.1:11434" >> docker.env +OLLAMA_HOST=0.0.0.0:11434 ollama serve +``` -# Or use IP address -echo "OLLAMA_HOST=http://$(ipconfig getifaddr en0):11434" >> docker.env # macOS +#### Solution C: Use the containerized Ollama instead +```bash +./start-docker.sh container + +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` + +This exports `OLLAMA_HOST=http://ollama:11434`, which wins over `docker.env` +because shell variables take precedence over `--env-file`. + +#### Solution D: Point at a specific host address +```bash +# Edit docker.env rather than appending duplicates - the last value wins, +# but duplicate keys make the file confusing. +# macOS example: +OLLAMA_HOST=http://$(ipconfig getifaddr en0):11434 ./start-docker.sh ``` -### 3. Container Build Failures +### 3. Backend / RAG API Wiring + +#### Problem: every chat fails with 502 "Could not connect to the RAG API server" + +The backend builds `/chat` and `/index` from `RAG_API_URL`. Inside compose that +must be `http://rag-api:8001` โ€” `localhost` there refers to the backend container. +Chat failures come back as JSON error bodies: 502 when the RAG API is unreachable, +504 when the call times out. + +```bash +docker compose exec backend env | grep RAG_API_URL +docker compose exec backend curl -s http://rag-api:8001/health +``` + +#### Problem: chat returns 504 "did not respond within 600s" + +The RAG API is single-threaded, so a chat request queued behind an indexing run can +exceed the timeout. Watch `docker compose logs -f rag-api`, and raise the limits if +your hardware is genuinely slow: + +```bash +RAG_API_TIMEOUT=1200 RAG_API_INDEX_TIMEOUT=7200 \ +docker compose --env-file docker.env up -d backend +``` + +#### Problem: the frontend calls the wrong host + +`NEXT_PUBLIC_API_URL` and `NEXT_PUBLIC_RAG_API_URL` are inlined by `next build`. +Setting them only in `environment:` cannot change an already-built image. + +```bash +NEXT_PUBLIC_API_URL=http://localhost:8000 \ +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 \ +docker compose --env-file docker.env up -d --build frontend +``` + +### 4. Container Build Failures #### Problem: Frontend build fails ``` @@ -138,39 +201,40 @@ ERROR: Failed to build frontend container #### Solution: Clean Build ```bash -# Stop containers ./start-docker.sh stop -# Clean Docker cache docker system prune -f docker builder prune -f -# Rebuild frontend only docker compose build --no-cache frontend docker compose up -d frontend -# Check logs docker compose logs frontend ``` +`Dockerfile.frontend` copies `package.json`, `package-lock.json`, `src/`, +`public/`, `next.config.ts`, `tsconfig.json`, `postcss.config.mjs` and +`eslint.config.mjs`. If you deleted one of those in a fork, the `COPY` fails. +`public/` is kept non-empty by `public/.gitkeep`. + #### Problem: Python package installation fails ``` ERROR: Could not install packages due to an EnvironmentError ``` -#### Solution: Update Dependencies +#### Solution ```bash -# Check requirements file exists +# Both Python images install requirements-docker.txt ls -la requirements-docker.txt -# Test package installation locally +# Test locally pip install -r requirements-docker.txt --dry-run -# Rebuild with updated base image +# Rebuild with an updated base image docker compose build --no-cache --pull rag-api ``` -### 4. Port Conflicts +### 5. Port Conflicts #### Problem: "Port already in use" ``` @@ -179,57 +243,55 @@ Error starting userland proxy: listen tcp4 0.0.0.0:3000: bind: address already i #### Solution: Find and Kill Conflicting Processes ```bash -# Check what's using the ports lsof -i :3000 -i :8000 -i :8001 -# Kill specific processes -pkill -f "npm run dev" # Frontend -pkill -f "server.py" # Backend -pkill -f "api_server" # RAG API +# A direct (non-Docker) run of the same stack is the usual culprit +python run_system.py --stop -# Or kill by port +# Or by port sudo kill -9 $(lsof -t -i:3000) sudo kill -9 $(lsof -t -i:8000) sudo kill -9 $(lsof -t -i:8001) -# Restart containers ./start-docker.sh ``` -### 5. Memory Issues +### 6. Memory Issues -#### Problem: Containers crash due to OOM (Out of Memory) +#### Problem: Containers crash due to OOM ``` Container killed due to memory limit ``` -#### Solution: Increase Docker Memory +The `rag-api` container holds the embedding model and the reranker in memory for +the life of the process, and enabling late chunking loads a second copy of the +embedder during indexing. + +#### Solution: Increase Docker Memory or shrink the models ```bash -# Check current memory usage docker stats --no-stream -# Increase Docker Desktop memory allocation # Docker Desktop โ†’ Settings โ†’ Resources โ†’ Memory โ†’ 8GB+ -# Monitor memory usage -docker stats +# Lighter configuration (rebuild indexes after changing the embedding model) +GENERATION_MODEL=qwen3.5:4b \ +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ +docker compose --env-file docker.env up -d rag-api backend -# Use smaller models if needed -ollama pull qwen3:0.6b # Instead of qwen3:8b +# Or switch the whole RAG API to the fast profile +RAG_CONFIG_MODE=fast docker compose --env-file docker.env up -d rag-api ``` #### Problem: System running slow ```bash -# Check host memory free -h # Linux vm_stat # macOS -# Clean up Docker resources docker system prune -f docker volume prune -f ``` -### 6. Volume Mount Issues +### 7. Volume Mount Issues #### Problem: Permission denied accessing files ``` @@ -238,18 +300,14 @@ Permission denied: /app/lancedb #### Solution: Fix Permissions ```bash -# Create directories if they don't exist mkdir -p lancedb index_store shared_uploads backend -# Fix permissions chmod -R 755 lancedb index_store shared_uploads chmod 664 backend/chat_data.db -# Check ownership ls -la lancedb/ shared_uploads/ backend/ -# Reset permissions if needed -sudo chown -R $USER:$USER lancedb shared_uploads backend +sudo chown -R $USER:$USER lancedb shared_uploads index_store backend ``` #### Problem: Database file not found @@ -257,24 +315,45 @@ sudo chown -R $USER:$USER lancedb shared_uploads backend No such file or directory: '/app/backend/chat_data.db' ``` -#### Solution: Initialize Database +`ChatDatabase` creates the parent directory and the file itself, so this normally +means the `./backend` bind mount is missing or `DB_PATH` points somewhere +unmounted. + ```bash -# Create empty database file -touch backend/chat_data.db +docker compose exec backend env | grep DB_PATH # expect /app/backend/chat_data.db +docker compose exec backend ls -la /app/backend -# Or initialize with schema +# Create it from the host if needed python -c " from backend.database import ChatDatabase db = ChatDatabase() -db.init_database() -print('Database initialized') +print('Database initialized at', db.db_path) " -# Restart containers ./start-docker.sh stop ./start-docker.sh ``` +#### Problem: Indexes exist in the sidebar but every answer says nothing was found + +The SQLite rows and the LanceDB tables must match. Restoring one without the other, +or changing `LANCEDB_PATH`, leaves index records pointing at tables that do not +exist. + +```bash +docker compose exec rag-api python -c " +import lancedb +print(lancedb.connect('/app/lancedb').table_names()) +" +docker compose exec backend sqlite3 /app/backend/chat_data.db \ + "select id, name, vector_table_name from indexes;" +``` + +#### Problem: `changing the embedding model requires rebuilding the index` + +Vector width is read from the loaded embedding model. Delete and rebuild the index +after changing `EMBEDDING_MODEL`. + --- ## ๐Ÿ” Advanced Debugging @@ -286,78 +365,80 @@ print('Database initialized') # RAG API container (most issues happen here) docker compose exec rag-api bash -# Check environment variables -docker compose exec rag-api env | grep -E "(OLLAMA|RAG|NODE)" +# Check the wiring +docker compose exec rag-api env | grep -E "OLLAMA|MODEL|DB_PATH|LANCEDB|RAG_CONFIG" # Test Python imports docker compose exec rag-api python -c " import sys print('Python version:', sys.version) -from rag_system.main import get_agent +from rag_system.factory import get_agent print('โœ… RAG system imports work') " # Backend container docker compose exec backend bash -python -c " +docker compose exec backend python -c " from backend.database import ChatDatabase print('โœ… Database imports work') " -# Frontend container -docker compose exec frontend sh -npm --version -node --version +# Frontend container (alpine -> sh, not bash) +docker compose exec frontend sh -c "node --version && npm --version" ``` #### Check Container Resources ```bash -# Monitor real-time resource usage docker stats -# Check individual container health docker compose ps docker inspect rag-api --format='{{.State.Health.Status}}' +docker inspect rag-api --format='{{json .State.Health}}' | head -c 2000 -# View container configurations docker compose config ``` #### Network Debugging + +The Python images are `python:3.11-slim` and the frontend is `node:20-alpine`; +**none of them ship `ping` or `nslookup`**. Use curl (installed in both Python +images) or busybox wget (in the alpine image): + ```bash -# Check network connectivity -docker compose exec rag-api ping backend -docker compose exec backend ping rag-api -docker compose exec rag-api ping host.docker.internal +# Backend โ†” RAG API by service name +docker compose exec backend curl -sv http://rag-api:8001/health +docker compose exec rag-api curl -sv http://backend:8000/health + +# Container โ†’ host Ollama +docker compose exec rag-api curl -sv http://host.docker.internal:11434/api/tags -# Check DNS resolution -docker compose exec rag-api nslookup host.docker.internal +# From the frontend (alpine) +docker compose exec frontend wget -qO- http://backend:8000/health -# Test HTTP connections -docker compose exec rag-api curl -v http://backend:8000/health -docker compose exec rag-api curl -v http://host.docker.internal:11434/api/tags +# If you really want ping/nslookup, install them in the running container first +docker compose exec rag-api sh -c "apt-get update && apt-get install -y iputils-ping dnsutils" ``` ### Log Analysis #### Container Logs ```bash -# View all logs ./start-docker.sh logs -# Follow specific service logs docker compose logs -f rag-api docker compose logs -f backend docker compose logs -f frontend -# Search for errors docker compose logs rag-api 2>&1 | grep -i error docker compose logs backend 2>&1 | grep -i "traceback\|error" -# Save logs to file docker compose logs > docker-debug.log 2>&1 ``` +The RAG API logs each stage of a query, so `docker compose logs -f rag-api` while +you ask a question shows exactly where time is going: retrieval, reranking, context +expansion, pruning, synthesis, verification. + #### System Logs ```bash # Docker daemon logs (Linux) @@ -373,72 +454,57 @@ journalctl -u docker.service -f ### Manual Container Testing -#### Test Individual Containers +`test_docker_build.sh` automates exactly this โ€” building each image and probing its +health endpoint: + ```bash -# Test RAG API alone -docker build -f Dockerfile.rag-api -t test-rag-api . -docker run --rm -p 8001:8001 -e OLLAMA_HOST=http://host.docker.internal:11434 test-rag-api & -sleep 30 -curl http://localhost:8001/models -pkill -f test-rag-api +./test_docker_build.sh +``` + +By hand: -# Test Backend alone +```bash +# RAG API alone +docker build -f Dockerfile.rag-api -t test-rag-api . +docker run -d --name test-rag-api -p 8001:8001 \ + -e OLLAMA_HOST=http://host.docker.internal:11434 \ + --add-host host.docker.internal:host-gateway test-rag-api +sleep 60 # models load before /health answers +curl http://localhost:8001/health +docker rm -f test-rag-api + +# Backend alone docker build -f Dockerfile.backend -t test-backend . -docker run --rm -p 8000:8000 test-backend & -sleep 30 +docker run -d --name test-backend -p 8000:8000 test-backend +sleep 15 curl http://localhost:8000/health -pkill -f test-backend +docker rm -f test-backend ``` -#### Integration Testing +### Integration Testing ```bash -# Full system test -./start-docker.sh +./start-docker.sh -y -# Wait for all services to be ready -sleep 60 +# Wait for the health-gated chain to come up +until curl -sf http://localhost:8000/health >/dev/null; do sleep 5; done -# Test complete workflow +# Create a session curl -X POST http://localhost:8000/sessions \ -H "Content-Type: application/json" \ -d '{"title": "Test Session"}' -# Test document upload (if you have a test PDF) -# curl -X POST http://localhost:8000/upload -F "file=@test.pdf" +# Create an index, upload a document, build it +IDX=$(curl -s -X POST http://localhost:8000/indexes \ + -H "Content-Type: application/json" \ + -d '{"name":"Smoke Test"}' | python3 -c "import sys,json;print(json.load(sys.stdin)['index_id'])") +curl -X POST "http://localhost:8000/indexes/$IDX/upload" -F "files=@test.pdf" +curl -X POST "http://localhost:8000/indexes/$IDX/build" \ + -H "Content-Type: application/json" -d '{"chunk_size": 512}' -# Clean up ./start-docker.sh stop ``` -### Automated Testing Script - -Create `test-docker-health.sh`: -```bash -#!/bin/bash -set -e - -echo "๐Ÿณ Docker Health Test Starting..." - -# Start containers -./start-docker.sh - -# Wait for services -echo "โณ Waiting for services to start..." -sleep 60 - -# Test endpoints -echo "๐Ÿ” Testing endpoints..." -curl -f http://localhost:3000 && echo "โœ… Frontend OK" || echo "โŒ Frontend FAIL" -curl -f http://localhost:8000/health && echo "โœ… Backend OK" || echo "โŒ Backend FAIL" -curl -f http://localhost:8001/models && echo "โœ… RAG API OK" || echo "โŒ RAG API FAIL" -curl -f http://localhost:11434/api/tags && echo "โœ… Ollama OK" || echo "โŒ Ollama FAIL" - -# Test container health -echo "๐Ÿ” Checking container health..." -docker compose ps - -echo "๐ŸŽ‰ Health test complete!" -``` +The upload form field must be named `files`. --- @@ -448,46 +514,38 @@ echo "๐ŸŽ‰ Health test complete!" #### Soft Reset ```bash -# Stop containers ./start-docker.sh stop - -# Clean up Docker resources docker system prune -f - -# Restart containers ./start-docker.sh ``` #### Hard Reset (โš ๏ธ Deletes all data) ```bash -# Stop everything ./start-docker.sh stop +docker compose down -# Remove all containers, images, and volumes -docker system prune -a --volumes - -# Remove local data (CAUTION: This deletes all your documents and chat history) -rm -rf lancedb/* shared_uploads/* backend/chat_data.db +# Application data lives in host directories, not volumes - +# `docker compose down -v` alone will NOT remove it. +rm -rf lancedb/* index_store/* shared_uploads/* backend/chat_data.db -# Rebuild from scratch +docker system prune -a --volumes ./start-docker.sh ``` #### Selective Reset - -Reset only specific components: ```bash -# Reset just the database +# Just the chat/index database ./start-docker.sh stop rm backend/chat_data.db ./start-docker.sh -# Reset just vector storage +# Just vector storage (indexes must be rebuilt; delete the DB rows too or they +# will point at tables that no longer exist) ./start-docker.sh stop rm -rf lancedb/* ./start-docker.sh -# Reset just uploaded documents +# Just uploaded documents rm -rf shared_uploads/* ``` @@ -497,31 +555,36 @@ rm -rf shared_uploads/* ### Resource Monitoring ```bash -# Monitor containers continuously watch -n 5 'docker stats --no-stream' -# Check disk usage docker system df -du -sh lancedb shared_uploads backend +du -sh lancedb shared_uploads index_store backend -# Monitor host resources htop # Linux -top # macOS/Windows +top # macOS ``` ### Performance Tuning ```bash -# Use smaller models for better performance -ollama pull qwen3:0.6b # Instead of qwen3:8b +# Fast profile: vector-only retrieval, no reranking, decomposition or verification +RAG_CONFIG_MODE=fast docker compose --env-file docker.env up -d rag-api -# Reduce Docker memory if needed -# Docker Desktop โ†’ Settings โ†’ Resources โ†’ Memory +# Smaller generation model +GENERATION_MODEL=qwen3.5:4b docker compose --env-file docker.env up -d rag-api backend -# Clean up regularly -docker system prune -f -docker volume prune -f +# Smaller embedding model (rebuild indexes afterwards) +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B docker compose --env-file docker.env up -d rag-api + +# Faster indexing: turn off contextual enrichment in the index build options +# (it makes one LLM call per chunk) ``` +Structural limits worth knowing before you tune: +- The RAG API is single-threaded โ€” requests are serialised. +- Indexing is synchronous; `POST /indexes/{id}/build` stays open until it finishes. +- Model weights are downloaded into the container's writable layer, so recreating + `rag-api` re-downloads them unless you mount a cache and set `HF_HOME`. + --- ## ๐Ÿ†˜ When All Else Fails @@ -530,29 +593,28 @@ docker volume prune -f #### 1. Direct Development (No Docker) ```bash -# Stop Docker containers ./start-docker.sh stop - -# Use direct development instead python run_system.py ``` #### 2. Minimal Docker (RAG API only) ```bash -# Run only RAG API in Docker docker build -f Dockerfile.rag-api -t rag-api . -docker run -p 8001:8001 rag-api +docker run -p 8001:8001 \ + -e OLLAMA_HOST=http://host.docker.internal:11434 \ + --add-host host.docker.internal:host-gateway rag-api -# Run other components directly -cd backend && python server.py & +# Run the rest directly, from the repository root +python backend/server.py & npm run dev ``` #### 3. Hybrid Approach ```bash -# Run some services in Docker, others directly -docker compose up -d rag-api -cd backend && python server.py & +docker compose --env-file docker.env up -d rag-api + +# Point the host backend at the container +RAG_API_URL=http://localhost:8001 python backend/server.py & npm run dev ``` @@ -560,27 +622,23 @@ npm run dev #### Diagnostic Information to Collect ```bash -# System information docker version docker compose version uname -a -# Container information docker compose ps docker compose config -# Resource information docker stats --no-stream docker system df -# Error logs docker compose logs > docker-errors.log 2>&1 ``` #### Support Channels 1. **Check GitHub Issues**: Search existing issues for similar problems -2. **Documentation**: Review the complete documentation in `Documentation/` -3. **Create Issue**: Include diagnostic information above +2. **Documentation**: Review `Documentation/` and `DOCKER_README.md` +3. **Create Issue**: Include the diagnostic information above --- @@ -590,8 +648,8 @@ Your Docker deployment is working correctly when: - โœ… `docker version` shows Docker is running - โœ… `curl http://localhost:11434/api/tags` shows Ollama is accessible -- โœ… `./start-docker.sh status` shows all containers healthy -- โœ… All health check URLs return 200 OK +- โœ… `docker compose ps` shows all services healthy +- โœ… `curl -f http://localhost:8000/health` and `.../8001/health` both return 200 - โœ… You can access the frontend at http://localhost:3000 - โœ… You can create document indexes successfully - โœ… You can chat with your documents @@ -601,4 +659,4 @@ Your Docker deployment is working correctly when: --- -**Still having issues?** Check the main `DOCKER_README.md` or create an issue with your diagnostic information. \ No newline at end of file +**Still having issues?** Check `DOCKER_README.md` or create an issue with your diagnostic information. diff --git a/Dockerfile.backend b/Dockerfile.backend index 9aeac269..3d67fbaa 100644 --- a/Dockerfile.backend +++ b/Dockerfile.backend @@ -3,9 +3,10 @@ FROM python:3.11-slim # Set working directory WORKDIR /app -# Install system dependencies +# Install system dependencies (sqlite3 CLI is used for database troubleshooting) RUN apt-get update && apt-get install -y \ curl \ + sqlite3 \ && rm -rf /var/lib/apt/lists/* # Copy requirements and install Python dependencies (using Docker-specific requirements) @@ -16,8 +17,8 @@ RUN pip install --no-cache-dir -r requirements.txt COPY backend/ ./backend/ COPY rag_system/ ./rag_system/ -# Create necessary directories and initialize database -RUN mkdir -p shared_uploads logs backend +# Create necessary directories +RUN mkdir -p shared_uploads index_store lancedb logs # Expose port EXPOSE 8000 @@ -26,6 +27,6 @@ EXPOSE 8000 HEALTHCHECK --interval=30s --timeout=10s --start-period=30s --retries=3 \ CMD curl -f http://localhost:8000/health || exit 1 -# Run the backend server -WORKDIR /app/backend -CMD ["python", "server.py"] \ No newline at end of file +# Run the backend server from the repo root so relative paths +# (shared_uploads/, index_store/, lancedb/) match local development +CMD ["python", "backend/server.py"] diff --git a/Dockerfile.frontend b/Dockerfile.frontend index 938d5c6c..39101163 100644 --- a/Dockerfile.frontend +++ b/Dockerfile.frontend @@ -1,4 +1,4 @@ -FROM node:18-alpine +FROM node:20-alpine # Set working directory WORKDIR /app @@ -12,20 +12,24 @@ COPY src/ ./src/ COPY public/ ./public/ COPY next.config.ts ./ COPY tsconfig.json ./ -COPY tailwind.config.js ./ COPY postcss.config.mjs ./ COPY eslint.config.mjs ./ -# Build the application (skip linting for Docker) -ENV NEXT_LINT=false +# NEXT_PUBLIC_* values are inlined by `next build`, so they must be set before it runs +ARG NEXT_PUBLIC_API_URL=http://localhost:8000 +ARG NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 +ENV NEXT_PUBLIC_API_URL=$NEXT_PUBLIC_API_URL +ENV NEXT_PUBLIC_RAG_API_URL=$NEXT_PUBLIC_RAG_API_URL + +# Build the application RUN npm run build # Expose port EXPOSE 3000 -# Health check +# Health check (node:20-alpine ships busybox wget, not curl) HEALTHCHECK --interval=30s --timeout=10s --start-period=60s --retries=3 \ - CMD curl -f http://localhost:3000 || exit 1 + CMD wget -qO- http://localhost:3000 >/dev/null 2>&1 || exit 1 # Start the application -CMD ["npm", "start"] \ No newline at end of file +CMD ["npm", "start"] diff --git a/Dockerfile.rag-api b/Dockerfile.rag-api index 940a2d0c..de9ea801 100644 --- a/Dockerfile.rag-api +++ b/Dockerfile.rag-api @@ -6,6 +6,7 @@ WORKDIR /app # Install system dependencies RUN apt-get update && apt-get install -y \ curl \ + sqlite3 \ build-essential \ && rm -rf /var/lib/apt/lists/* @@ -23,9 +24,13 @@ RUN mkdir -p lancedb index_store shared_uploads logs # Expose port EXPOSE 8001 -# Health check -HEALTHCHECK --interval=30s --timeout=10s --start-period=60s --retries=3 \ - CMD curl -f http://localhost:8001/models || exit 1 +# Health check. The RAG API is deliberately single-threaded (its shared +# in-memory pipeline/agent state is only safe because one request runs at a +# time), so one long /chat or /index blocks /health. The generous start_period +# covers model loading, and 10 retries ride out routine long requests instead +# of flapping the container to unhealthy. +HEALTHCHECK --interval=30s --timeout=10s --start-period=120s --retries=10 \ + CMD curl -f http://localhost:8001/health || exit 1 # Run the RAG API server -CMD ["python", "-m", "rag_system.api_server"] \ No newline at end of file +CMD ["python", "-m", "rag_system.api_server"] diff --git a/Documentation/api_reference.md b/Documentation/api_reference.md index 1e995bdc..06fa24be 100644 --- a/Documentation/api_reference.md +++ b/Documentation/api_reference.md @@ -1,161 +1,370 @@ -# ๐Ÿ“š API Reference (Backend & RAG API) +# ๐Ÿ“š API Reference (Backend Gateway & RAG API) -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ + +Two HTTP services, both reachable from the browser: + +| Service | Base URL | Source | +|---------|----------|--------| +| Backend gateway | `http://localhost:8000` | `backend/server.py` | +| RAG API | `http://localhost:8001` | `rag_system/api_server.py` | + +Both send `Access-Control-Allow-Origin: *` on every response (including 404s, which return JSON bodies) and answer `OPTIONS` preflights. Neither sends `Access-Control-Allow-Credentials`, and neither implements authentication. + +**Wire format** is snake_case. Both services additionally accept camelCase and normalise it to the same canonical key at parse time, so `rerankerTopK` and `reranker_top_k` are interchangeable. When both spellings are present, the explicit snake_case value wins. The gateway uses an explicit alias table that covers every option it accepts (plus a few legacy names such as `latechunk` and `decompose`); the RAG API converts any camelCase key generically. --- -## Backend HTTP API (Python `backend/server.py`) -**Base URL**: `http://localhost:8000` +## 1. Backend Gateway โ€” `http://localhost:8000` + +### 1.1 Route table -| Endpoint | Method | Description | Request Body | Success Response | +| Endpoint | Method | Description | Request body | Success response | |----------|--------|-------------|--------------|------------------| -| `/health` | GET | Health probe incl. Ollama status & DB stats | โ€“ | 200 JSON `{ status, ollama_running, available_models, database_stats }` | -| `/chat` | POST | Stateless chat (no session) | `{ message:str, model?:str, conversation_history?:[{role,content}]}` | 200 `{ response:str, model:str, message_count:int }` | -| `/sessions` | GET | List all sessions | โ€“ | `{ sessions:ChatSession[], total:int }` | -| `/sessions` | POST | Create session | `{ title?:str, model?:str }` | 201 `{ session:ChatSession, session_id }` | -| `/sessions/` | GET | Get session + msgs | โ€“ | `{ session, messages }` | -| `/sessions/` | DELETE | Delete session | โ€“ | `{ message, deleted_session_id }` | -| `/sessions//rename` | POST | Rename session | `{ title:str }` | `{ message, session }` | -| `/sessions//messages` | POST | Session chat (builds history) | See ChatRequest + retrieval opts โ–ผ | `{ response, session, user_message_id, ai_message_id }` | -| `/sessions//documents` | GET | List uploaded docs | โ€“ | `{ files:string[], file_count:int, session }` | -| `/sessions//upload` | POST multipart | Upload docs to session | field `files[]` | `{ message, uploaded_files, processing_results?, session_documents?, total_session_documents? }` | -| `/sessions//index` | POST | Trigger RAG indexing for session | `{ latechunk?, doclingChunk?, chunkSize?, ... }` | `{ message }` | -| `/sessions//indexes` | GET | List indexes linked to session | โ€“ | `{ indexes, total }` | -| `/sessions//indexes/` | POST | Link index to session | โ€“ | `{ message }` | -| `/sessions/cleanup` | GET | Remove empty sessions | โ€“ | `{ message, cleanup_count }` | -| `/models` | GET | List generation / embedding models | โ€“ | `{ generation_models:str[], embedding_models:str[] }` | +| `/health` | GET | Health probe with Ollama status and DB stats | โ€“ | `{ status, ollama_running, available_models, database_stats }` | +| `/chat` | POST | Stateless chat, no session, no retrieval | `{ message, model?, conversation_history? }` | `{ response, model, message_count }` | +| `/models` | GET | Available generation / embedding models | โ€“ | `{ generation_models, embedding_models }` | +| `/sessions` | GET | List sessions | โ€“ | `{ sessions, total }` | +| `/sessions` | POST | Create a session | `{ title?, model? }` | 201 `{ session, session_id }` | +| `/sessions/cleanup` | GET | Delete sessions that have no messages | โ€“ | `{ message, cleanup_count }` | +| `/sessions/` | GET | Session plus its messages | โ€“ | `{ session, messages }` | +| `/sessions/` | DELETE | Delete a session and its messages | โ€“ | `{ deleted: true }` | +| `/sessions//rename` | POST | Rename a session | `{ title }` | `{ message, session }` | +| `/sessions//messages` | POST | Session chat (persisted) | [Session chat request](#12-session-chat-request) | `{ response, session, source_documents, used_rag }` | +| `/sessions//messages/save` | POST | Persist a completed streamed turn (the browser calls this after `POST :8001/chat/stream` finishes) | `{ user_message, assistant_message, source_documents?, steps? }` | `{ session, user_message_id, ai_message_id }` โ€” sources and steps are stored in the assistant message's `metadata.source_documents` / `metadata.steps` | +| `/sessions//documents` | GET | Files uploaded to a session | โ€“ | `{ session, files, file_count }` | +| `/sessions//upload` | POST | Upload files to a session | multipart, field `files` | `{ message, uploaded_files }` | +| `/sessions//index` | POST | Index the session's documents | [Index options](#14-index-options) (optional) | the RAG API `/index` response | +| `/sessions//indexes` | GET | Indexes linked to a session | โ€“ | `{ indexes, total }` | +| `/sessions//indexes/` | POST | Link an index to a session | โ€“ | `{ message }` | | `/indexes` | GET | List all indexes | โ€“ | `{ indexes, total }` | -| `/indexes` | POST | Create index | `{ name:str, description?:str, metadata?:dict }` | `{ index_id }` | -| `/indexes/` | GET | Get single index | โ€“ | `{ index }` | -| `/indexes/` | DELETE | Delete index | โ€“ | `{ message, index_id }` | -| `/indexes//upload` | POST multipart | Upload docs to index | field `files[]` | `{ message, uploaded_files }` | -| `/indexes//build` | POST | Build / rebuild index (RAG) | `{ latechunk?, doclingChunk?, ...}` | 200 `{ response?, message?}` (idempotent) | +| `/indexes` | POST | Create a named index | `{ name, description?, metadata? }` | 201 `{ index_id }` | +| `/indexes/` | GET | One index | โ€“ | the index object (see below) | +| `/indexes/` | DELETE | Delete an index, its links and its LanceDB table | โ€“ | `{ message, index_id }` | +| `/indexes//upload` | POST | Upload files to an index | multipart, field `files` | `{ message, uploaded_files }` | +| `/indexes//build` | POST | Build / rebuild the index | [Index options](#14-index-options) (optional) | `{ response, ...echoed options }` | + +`uploaded_files` entries are `{ filename, stored_path }`. The index object returned by `GET /indexes/` is `{ id, name, description, created_at, updated_at, vector_table_name, metadata, documents[] }` โ€” it is **not** wrapped in `{ index: โ€ฆ }`. + +> **Body required.** `POST /chat`, `POST /sessions`, `POST /sessions//messages` and `POST /indexes` read `Content-Length` unconditionally; send a JSON body (at minimum `{}`) or the request fails with a 500. `POST /sessions//rename` returns a clean `400 { "error": "Request body required" }`. `POST /sessions//index` and `POST /indexes//build` treat the body as optional. + +### 1.2 Session chat request + +`POST /sessions//messages` + +```jsonc +{ + "message": "string", // required + "model": "qwen3.5:9b", // optional โ€“ generation model for this request + "force_rag": false, // optional โ€“ skip gateway routing, always call the RAG API + + // Retrieval options, forwarded to the RAG API when the RAG route is taken. + // Omitted options are not forwarded, so the pipeline profile default applies. + "compose_sub_answers": true, + "query_decompose": true, // alias: "decompose" + "ai_rerank": true, // profile default when omitted is ON (arm G, 2026-08-14; eval/DECISIONS.md) + "context_expand": true, + "verify": true, + "retrieval_k": 20, + "context_window_size": 1, + "reranker_top_k": 10, + "retrieval_mode": "hybrid", // "hybrid" | "vector_only" | "fts_only" (alias: "search_type") + "provence_prune": false, + "provence_threshold": 0.1, + "filters": { } // optional metadata filter object, forwarded verbatim and + // validated by the RAG API (ยง2.1). A present filter also + // forces the RAG route, like force_rag. +} +``` + +camelCase aliases are accepted for all of them (`composeSubAnswers`, `queryDecompose`, `aiRerank`, `contextExpand`, `retrievalK`, `contextWindowSize`, `rerankerTopK`, `retrievalMode`, `searchType`, `provencePrune`, `provenceThreshold`, `forceRag`). + +Response: + +```jsonc +{ + "response": "string", // assistant answer, tags stripped + "session": { /* ChatSession */ }, + "source_documents": [], // empty on the direct-LLM route + "used_rag": true +} +``` + +This endpoint persists the turn itself: it writes the user message, derives a session title from the first message, then writes the assistant message with its sources in `metadata.source_documents`. (Streamed turns are persisted separately via `/messages/save`.) + +`force_rag` decides gateway routing **and** is forwarded to the RAG API, so the agent's own triage is skipped too. + +### 1.3 Gateway routing + +`POST /sessions//messages` picks its path with a deterministic gate โ€” no model call, no retrieval, sub-millisecond: + +1. `force_rag: true` (or `forceRag`) โ†’ RAG, unconditionally. +2. Session has no linked indexes โ†’ answered directly by Ollama, no retrieval. +3. The whole message is smalltalk (`hello`, `thanks!`, `bye`, `ok` โ€” an allowlist regex capped at six words) or a question about the assistant itself (`who are you`, `what model are you`) โ†’ direct. +4. Anything else โ†’ RAG. + +`used_rag` in the response tells you which way it went. The gate errs toward RAG on purpose: the RAG API's own agent triage still runs on every forwarded request and can answer directly without retrieving, so a false "use RAG" costs one triage call, not a wrong answer. If you need the answer grounded in documents regardless, send `force_rag`. + +The streaming path (`:8001/chat/stream`, the UI default) never touches this gate โ€” the browser calls the RAG API directly and only the agent triage applies. + +### 1.4 Index options + +Accepted by `POST /sessions//index` and `POST /indexes//build`, normalised and forwarded to the RAG API `/index`. **Options you omit are not sent**, so the RAG API's own defaults apply (ยง2.5). + +```jsonc +{ + "chunk_size": 512, + "window_size": 2, + "retrieval_mode": "hybrid", // "hybrid" | "vector_only" | "fts_only" + "enable_enrich": true, + "enable_latechunk": false, // aliases: "latechunk", "enableLatechunk" + "enable_docling_chunk": true, // aliases: "doclingChunk", "enableDoclingChunk"; false = legacy chunker โ€” see ยง2.5 + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", // aliases: "overviewModel", "overview_model" + "batch_size_embed": 50, + "batch_size_enrich": 25 +} +``` + +`POST /sessions//index` returns the RAG API's `/index` response unchanged, or `200 { "message": "No documents to index for this session." }` when the session has no uploaded files. + +`POST /indexes//build` echoes the canonical options back alongside the RAG API response, renaming `enable_latechunk` โ†’ `latechunk`, `enable_docling_chunk` โ†’ `docling_chunk` and `overview_model_name` โ†’ `overview_model`, and stores the same values in the index's metadata. If the RAG API reports that the table already exists, the build is treated as idempotent and returns `{ message: "Index already built โ€“ skipping rebuild.", note }`. + +### 1.5 Error responses + +| Status | When | +|--------|------| +| 400 | Missing required field, invalid JSON, no files in a multipart upload | +| 404 | Unknown route, unknown session, unknown index | +| 500 | Unhandled server error | +| 502 | Could not connect to the RAG API (`RAG_API_URL`), or the RAG API returned a non-200 | +| 503 | Ollama is not reachable (`POST /chat` only) | +| 504 | The RAG API did not answer within `RAG_API_TIMEOUT` (chat, default 600 s) or `RAG_API_INDEX_TIMEOUT` (indexing, default 3600 s) | + +Error bodies are JSON: `{ "error": "..." }`. --- -## RAG API (Python `rag_system/api_server.py`) -**Base URL**: `http://localhost:8001` +## 2. RAG API โ€” `http://localhost:8001` -| Endpoint | Method | Description | Request Body | Success Response | +| Endpoint | Method | Description | Request body | Success response | |----------|--------|-------------|--------------|------------------| -| `/chat` | POST | Run RAG query with full pipeline | See RAG ChatRequest โ–ผ | `{ answer:str, source_documents:[], reasoning?:str, confidence?:float }` | -| `/chat/stream` | POST | Run RAG query with SSE streaming | Same as /chat | Server-Sent Events stream | -| `/index` | POST | Index documents with full configuration | See Index Request โ–ผ | `{ message:str, indexed_files:[], table_name:str }` | -| `/models` | GET | List available models | โ€“ | `{ generation_models:str[], embedding_models:str[] }` | +| `/health` | GET | Liveness probe | โ€“ | `{ "status": "ok" }` | +| `/models` | GET | Models available to the active LLM backend | โ€“ | `{ generation_models, embedding_models }` | +| `/chat` | POST | Run the full agent pipeline | [Chat request](#21-chat-request) | `{ answer, source_documents }` | +| `/chat/stream` | POST | Same pipeline, streamed as SSE | [Chat request](#21-chat-request) | `text/event-stream` | +| `/index` | POST | Index documents | [Index request](#25-index-request) | see ยง2.6 | + +Unknown routes return 404 `{ "error": "Not Found" }`. + +### 2.1 Chat request -### RAG ChatRequest (Advanced Options) ```jsonc { - "query": "string", // Required โ€“ user question - "session_id": "string", // Optional โ€“ for session context - "table_name": "string", // Optional โ€“ specific index table - "compose_sub_answers": true, // Optional โ€“ compose sub-answers - "query_decompose": true, // Optional โ€“ decompose complex queries - "ai_rerank": false, // Optional โ€“ AI-powered reranking - "context_expand": false, // Optional โ€“ context expansion - "verify": true, // Optional โ€“ answer verification - "retrieval_k": 20, // Optional โ€“ number of chunks to retrieve - "context_window_size": 1, // Optional โ€“ context window size - "reranker_top_k": 10, // Optional โ€“ top-k after reranking - "search_type": "hybrid", // Optional โ€“ "hybrid|dense|fts" - "dense_weight": 0.7, // Optional โ€“ dense search weight (0-1) - "force_rag": false, // Optional โ€“ bypass triage, force RAG - "provence_prune": false, // Optional โ€“ sentence-level pruning - "provence_threshold": 0.8, // Optional โ€“ pruning threshold - "model": "qwen3:8b" // Optional โ€“ generation model override + "query": "string", // required + "session_id": "string", // optional โ€“ loads the session's overviews and index metadata + "table_name": "string", // optional โ€“ LanceDB table; otherwise resolved from session_id + "model": "qwen3.5:9b", // optional โ€“ generation model for this request only + + "compose_sub_answers": true, // optional โ€“ profile default when omitted + "query_decompose": true, // optional โ€“ profile default when omitted + "ai_rerank": true, // optional โ€“ profile default when omitted, which is ON (arm G, 2026-08-14; eval/DECISIONS.md) + "context_expand": true, // optional โ€“ false forces a context window of 0 + "verify": true, // optional โ€“ profile default when omitted + "force_rag": false, // optional โ€“ skip triage and go straight to retrieval + + "retrieval_k": 20, // optional โ€“ profile default when omitted + "context_window_size": 1, // optional โ€“ profile default when omitted + "reranker_top_k": 10, // optional โ€“ profile default when omitted + + "retrieval_mode": "hybrid", // "hybrid" | "vector_only" | "fts_only"; alias "search_type" + "provence_prune": false, // optional โ€“ sentence-level pruning + "provence_threshold": 0.1, // optional โ€“ pruning threshold + + "filters": { // optional โ€“ metadata filter (roadmap 4.4); prefilters BOTH search legs + "document_id": { "eq": "07_nda.pdf" }, // also: {"in": [...]}, {"contains": "..."} + "chunk_index": { "gte": 0, "lt": 10 } // also on document_name (contains), chunk_id (eq/in) + } } ``` -### Index Request (Document Indexing) +Notes: + +* Every option, including `retrieval_k`, `context_window_size` and `reranker_top_k`, falls back to the pipeline profile when omitted; only options you actually send override the profile, and the override is scoped to that request โ€” the agent snapshots its config before applying request options and restores it afterwards. +* An unsupported `retrieval_mode` is rejected with `400 { "error": "Unsupported retrieval mode 'โ€ฆ'. Supported: hybrid, vector_only, fts_only." }`. +* `model` is applied for the duration of the request and then restored. A model id that does not match the active `LLM_BACKEND` (for example an Ollama tag while `LLM_BACKEND=watsonx`) is ignored with a warning. +* If the table's index metadata records an `embedding_model`, the retrieval pipeline switches to it before searching. +* `filters` is validated by `rag_system/retrieval/filters.py`: unknown fields/operators, wrong types, empty IN-lists, an empty object, and values containing quoting characters (`'`, `"`, `\`, `;`, backtick, control chars โ€” refused, never escaped) all return `400` with the validator's message. Sending `filters` also skips triage, like `force_rag`. Page/date filtering is not supported (the values live inside the metadata JSON column). + +### 2.2 Chat response + ```jsonc { - "file_paths": ["path1.pdf", "path2.pdf"], // Required โ€“ files to index - "session_id": "string", // Required โ€“ session identifier - "chunk_size": 512, // Optional โ€“ chunk size (default: 512) - "chunk_overlap": 64, // Optional โ€“ chunk overlap (default: 64) - "enable_latechunk": true, // Optional โ€“ enable late chunking - "enable_docling_chunk": false, // Optional โ€“ enable DocLing chunking - "retrieval_mode": "hybrid", // Optional โ€“ "hybrid|dense|fts" - "window_size": 2, // Optional โ€“ context window - "enable_enrich": true, // Optional โ€“ enable enrichment - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", // Optional โ€“ embedding model - "enrich_model": "qwen3:0.6b", // Optional โ€“ enrichment model - "overview_model_name": "qwen3:0.6b", // Optional โ€“ overview model - "batch_size_embed": 50, // Optional โ€“ embedding batch size - "batch_size_enrich": 25 // Optional โ€“ enrichment batch size + "answer": "string", + "source_documents": [ + { + "chunk_id": "string", + "text": "string", + "score": 0.0164, // higher is better; RRF score in hybrid mode + "document_id": "report.pdf", + "chunk_index": 12, + "metadata": { }, + "rerank_score": 0.87, // present only when reranking ran + "bm25": 4.21 // present only when the full-text leg matched this chunk + } + ], + "token_usage": { // per-query token accounting (roadmap 4.5, always on) + "by_stage": { "synthesis": { "prompt_tokens": 1192, "output_tokens": 861, "calls": 1 } }, + "total": { "prompt_tokens": 1192, "output_tokens": 861, "calls": 1, "total_tokens": 2053 } + }, + "document_escalation": [ ] // present only when full-document escalation fired (flag-gated, off by default) } ``` -> **Note on CORS** โ€“ All endpoints include `Access-Control-Allow-Origin: *` header. +An absent key in `by_stage` means that stage made no LLM call, not that it cost +zero tokens. Only Ollama reports real counts; watsonx reports zeros. ---- +There is no top-level `confidence` or `reasoning` field. When verification is enabled the confidence is appended to `answer` as `" [Confidence: N%]"`, plus `" [Warning: Low confidence. Groundedness: ]"` when the answer is judged ungrounded or scores below 50. Nothing is appended when the verifier's score parses as 0. -## Frontend Wrapper (`src/lib/api.ts`) -The React/Next.js frontend calls the backend via a typed wrapper. Important methods & payloads: - -| Method | Backend Endpoint | Payload Shape | -|--------|------------------|---------------| -| `checkHealth()` | `/health` | โ€“ | -| `sendMessage({ message, model?, conversation_history? })` | `/chat` | ChatRequest | -| `getSessions()` | `/sessions` | โ€“ | -| `createSession(title?, model?)` | `/sessions` | โ€“ | -| `getSession(sessionId)` | `/sessions/` | โ€“ | -| `sendSessionMessage(sessionId, message, opts)` | `/sessions//messages` | `ChatRequest + retrieval opts` | -| `uploadFiles(sessionId, files[])` | `/sessions//upload` | multipart | -| `indexDocuments(sessionId)` | `/sessions//index` | opts similar to buildIndex | -| `buildIndex(indexId, opts)` | `/indexes//build` | Index build options | -| `linkIndexToSession` | `/sessions//indexes/` | โ€“ | +When nothing is retrieved, `answer` is `"I could not find an answer in the documents."` and `source_documents` is empty. ---- +### 2.3 `POST /chat/stream` (SSE) + +Same request body. The response is `text/event-stream`; each event is a single line: + +``` +data: {"type": "", "data": } +``` -## Payload Definitions (Canonical) +| Event | Payload | Emitted when | +|-------|---------|--------------| +| `analyze` | `{query}` | Start of the run | +| `direct_answer` | `{}` | Triage chose a direct answer | +| `decomposition` | `{sub_queries}` | Query decomposition produced sub-queries | +| `retrieval_started` | `{mode}` or `{count}` | `{mode}` from the retrieval pipeline, `{count}` from the decomposition branch | +| `retrieval_done` | `{count}` | Retrieval finished | +| `retrieval_retry` | `{score_before, score_after, kept, โ€ฆ}` | The evidence-sufficiency retry ran (design_rationale ยง5) | +| `crossref_hop` | `{targets, chunks_added}` | The cross-reference hop pulled chunks (flag-gated, off by default) | +| `document_escalation` | `{document_id, document_name, chunks_used, chunks_total, approx_tokens, truncated, signal, score, threshold, token_budget}` | Full-document escalation fired (flag-gated, off by default); never contains the document text | +| `rerank_started` / `rerank_done` | `{count}` | Reranking | +| `context_expand_started` / `context_expand_done` | `{count}` | Context expansion | +| `prune_started` / `prune_done` | `{count}` | Provence pruning (only when enabled) | +| `token` | `{text}` | Streamed answer tokens | +| `sub_query_token` | `{index, text, question}` | Tokens from a parallel sub-query | +| `sub_query_result` | `{index, query, answer, source_documents}` | A sub-query finished | +| `single_query_result` | the pipeline result | Decomposition produced exactly one sub-query | +| `final_answer` | `{answer, source_documents}` | Composed answer ready | +| `complete` | `{answer, source_documents, token_usage}` | Final event; clients may close here. `token_usage` has the shape shown in ยง2.2 | +| `error` | `{error}` | Failure after the stream opened | + +> This endpoint does **not** write to SQLite. The browser's default chat path calls it directly and, once the `complete` event arrives, persists the finished turn via `POST :8000/sessions//messages/save`. Clients that consume the stream directly must do the same if they want the turn in the session history. + +### 2.4 `GET /models` -### ChatRequest (frontend โ‡„ backend) ```jsonc { - "message": "string", // Required โ€“ raw user text - "model": "string", // Optional โ€“ generation model id - "conversation_history": [ // Optional โ€“ prior turn list - { "role": "user|assistant", "content": "string" } - ] + "generation_models": ["qwen3.5:4b", "qwen3.5:9b"], + "embedding_models": ["Qwen/Qwen3-Embedding-0.6B", "Qwen/Qwen3-Embedding-4B", "Qwen/Qwen3-Embedding-8B", "microsoft/harrier-oss-v1-0.6b"] } ``` -### Session Chat Extended Options +With `LLM_BACKEND=ollama` the generation list comes from `GET {OLLAMA_HOST}/api/tags` (5 s timeout; on failure the list is simply shorter), split by a substring match on `embed` / `bge` / `embedding`. With `LLM_BACKEND=watsonx` it lists the configured WatsonX generation and enrichment models. The embedding list always includes the pipeline's configured embedding model, the default `microsoft/harrier-oss-v1-0.6b` and the three Qwen3-Embedding sizes; it is returned sorted and de-duplicated. + +### 2.5 Index request + ```jsonc { - "composeSubAnswers": true, - "decompose": true, - "aiRerank": false, - "contextExpand": false, - "verify": true, - "retrievalK": 10, - "contextWindowSize": 5, - "rerankerTopK": 20, - "searchType": "fts|hybrid|dense", - "denseWeight": 0.75, - "force_rag": false + "file_paths": ["/abs/path/a.pdf", "/abs/path/b.docx"], // required + "session_id": "string", // optional โ€“ overview file + metadata target + "table_name": "string", // optional โ€“ otherwise resolved from session_id + + "chunk_size": 512, // default 512 โ€“ token budget per chunk + "window_size": 2, // default 2 โ€“ contextual-enrichment window + "enable_enrich": true, // default true + "enable_latechunk": false, // default false + "enable_docling_chunk": true, // default true โ€“ false selects the legacy fixed-size chunker + "retrieval_mode": "hybrid", // optional โ€“ validated, recorded on the index config + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", + "batch_size_embed": 50, // default 50 + "batch_size_enrich": 25 // default 25 } ``` -### Index Build Options +* `session_id` is **optional**. Without it the default LanceDB table is used and overviews go to the global `index_store/overviews/overviews.jsonl` instead of a per-index file. +* `retrieval_mode` cannot change the artifacts written at index time; it is validated (400 on an unsupported value) and stored in the index config as `retrieval.search_type`. The mode that matters is the one you send at query time. +* `embedding_model` is applied to this build **and** recorded in the index metadata when the `session_id` names an actual index (checked by existence); session-scoped builds โ€” whose id is a chat-session UUID with no indexes row โ€” skip the recording. This is what makes queries against that index use the same embedder. +* `enable_latechunk` defaults to `false` here, so an HTTP build without the flag writes no `_lc` table even though the `default` profile enables late chunking. +* `enable_docling_chunk` defaults to `true`; sending `false` selects the legacy fixed-size chunker instead of Docling's structure-aware chunker. +* Unknown fields are ignored silently. + +### 2.6 Index response + ```jsonc { - "latechunk": true, - "doclingChunk": false, - "chunkSize": 512, - "chunkOverlap": 64, - "retrievalMode": "hybrid|dense|fts", - "windowSize": 2, - "enableEnrich": true, - "embeddingModel": "Qwen/Qwen3-Embedding-0.6B", - "enrichModel": "qwen3:0.6b", - "overviewModel": "qwen3:0.6b", - "batchSizeEmbed": 64, - "batchSizeEnrich": 32 + "message": "Indexing process for 2 file(s) completed successfully.", + "table_name": "text_pages_", + "latechunk": false, + "docling_chunk": true, + "indexing_config": { + "chunk_size": 512, + "retrieval_mode": "hybrid", + "window_size": 2, + "enable_enrich": true, + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", + "batch_size_embed": 50, + "batch_size_enrich": 25 + } } ``` +`indexing_config.embedding_model` reports the model actually used for the build, not the raw request value. There is no `indexed_files` field. + +Errors: `400 { "error": "A 'file_paths' list is required." }`, `400 { "error": "Invalid JSON" }`, `400` for an unsupported `retrieval_mode`, and `500 { "error": "Failed to start indexing: โ€ฆ" }`. + +--- + +## 3. Frontend Wrapper (`src/lib/api.ts`) + +The typed client exported as `chatAPI`. Base URLs come from `NEXT_PUBLIC_API_URL` (default `http://localhost:8000`) and `NEXT_PUBLIC_RAG_API_URL` (default `http://localhost:8001`), both inlined at build time. **The browser talks to both origins**, so a deployment must expose both. + +| Method | Target | +|--------|--------| +| `checkHealth()` | `GET :8000/health` | +| `sendMessage({message, model?, conversation_history?})` | `POST :8000/chat` | +| `getSessions()` | `GET :8000/sessions` | +| `createSession(title?, model?)` | `POST :8000/sessions` | +| `getSession(sessionId)` | `GET :8000/sessions/` | +| `sendSessionMessage(sessionId, message, opts)` | `POST :8000/sessions//messages` | +| `saveStreamedTurn(sessionId, userMessage, assistantMessage, sourceDocuments?)` | `POST :8000/sessions//messages/save` | +| `deleteSession(sessionId)` | `DELETE :8000/sessions/` | +| `renameSession(sessionId, title)` | `POST :8000/sessions//rename` | +| `uploadFiles(sessionId, files)` | `POST :8000/sessions//upload` | +| `indexDocuments(sessionId)` | `POST :8000/sessions//index` (no options) | +| `getModels()` | `GET :8000/models` | +| `createIndex(name, description?, metadata?)` | `POST :8000/indexes` | +| `uploadFilesToIndex(indexId, files)` | `POST :8000/indexes//upload` | +| `buildIndex(indexId, opts)` | `POST :8000/indexes//build` | +| `listIndexes()` | `GET :8000/indexes` | +| `getSessionIndexes(sessionId)` | `GET :8000/sessions//indexes` | +| `deleteIndex(indexId)` | `DELETE :8000/indexes/` | +| `linkIndexToSession(sessionId, indexId)` | `POST :8000/sessions//indexes/` | +| **`streamSessionMessage(params, onEvent)`** | **`POST :8001/chat/stream`** โ€” the default chat path | + +The camelCase argument names on these methods (`retrievalK`, `rerankerTopK`, `doclingChunk`, โ€ฆ) are TypeScript parameter names. `sendSessionMessage` and `streamSessionMessage` serialise them to snake_case JSON; `buildIndex` sends camelCase, which both servers normalise. + +Exported model defaults, kept in step with `rag_system/main.py`: + +```ts +export const DEFAULT_GENERATION_MODEL = 'qwen3.5:9b'; +export const DEFAULT_ENRICHMENT_MODEL = 'qwen3.5:4b'; +export const DEFAULT_EMBEDDING_MODEL = 'microsoft/harrier-oss-v1-0.6b'; +``` + --- -_This reference is derived from static code analysis of `backend/server.py`, `rag_system/api_server.py`, and `src/lib/api.ts`. Keep it in sync with route or type changes._ \ No newline at end of file +_Derived by reading `backend/server.py`, `rag_system/api_server.py` and `src/lib/api.ts`. Keep it in sync with route, option and response-shape changes._ diff --git a/Documentation/architecture_overview.md b/Documentation/architecture_overview.md index 02bb4ef6..595d18f5 100644 --- a/Documentation/architecture_overview.md +++ b/Documentation/architecture_overview.md @@ -1,83 +1,206 @@ # ๐Ÿ—๏ธ System Architecture Overview -_Last updated: 2025-07-06_ +_Last updated: 2026-08-08_ -This document explains how data and control flow through the Advanced **RAG System** โ€” from a user's browser all the way to model inference and back. It is intended as the **ground-truth reference** for engineers and integrators. +This document explains how data and control flow through **localGPT** โ€” from a user's browser to model inference and back. It is the ground-truth reference for engineers and integrators; every port, path and endpoint below was read out of the current source. --- -## 1. Bird's-Eye Diagram +## 1. Process Topology + +The system is four separate OS processes. Nothing in `rag_system` runs inside the backend gateway โ€” the gateway talks to the RAG API over HTTP. ```mermaid flowchart LR subgraph Client - U["๐Ÿ‘ค User (Browser)"] - FE["Next.js Front-end\nReact Components"] + U["๐Ÿ‘ค User (Browser)"] + FE["Next.js frontend
:3000"] U --> FE end - subgraph Network - FE -->|HTTP/JSON| BE["Python HTTP Server\nbackend/server.py"] - end - - subgraph Core["rag_system core package"] - BE --> LOOP["Agent Loop\n(rag_system/agent/loop.py)"] - BE --> IDX["Indexing Pipeline\n(pipelines/indexing_pipeline.py)"] - - LOOP --> RP["Retrieval Pipeline\n(pipelines/retrieval_pipeline.py)"] - LOOP --> VER["Verifier (Grounding Check)"] - RP --> RET["Retrievers\nBM25 | Dense | Hybrid"] - RP --> RER["AI Reranker"] - RP --> SYNT["Answer Synthesiser"] + subgraph Services + BE["Backend gateway
backend/server.py
:8000"] + RAG["RAG API
rag_system/api_server.py
:8001"] + OL["Ollama
:11434"] end subgraph Storage - LDB[("LanceDB Vector Tables")] - SQL[("SQLite โ€“ chat & metadata")] + SQL[("SQLite
backend/chat_data.db")] + LDB[("LanceDB
./lancedb")] + FS["File system
shared_uploads/ ยท index_store/"] end - subgraph Models - OLLAMA["Ollama Server\n(qwen3, etc.)"] - HF["HuggingFace Hosted\nEmbedding/Reranker Models"] + FE -->|"REST: sessions, indexes, uploads, non-streaming chat"| BE + FE -->|"POST /chat/stream (SSE) โ€” default chat path"| RAG + BE -->|"POST /chat, POST /index"| RAG + BE -->|"direct answers only (routing is local)"| OL + RAG -->|"generation, enrichment, verification"| OL + + BE -->|"sessions, messages, indexes"| SQL + RAG -->|"index metadata only"| SQL + RAG --> LDB + RAG --> FS +``` + +| Process | Entry point | Port | Server type | +|---------|-------------|------|-------------| +| Frontend | `npm run dev` (or `npm run build && npm run start`) | 3000 | Next.js 15 / React 19 | +| Backend gateway | `python backend/server.py` | 8000 | `socketserver.ThreadingTCPServer` (concurrent) | +| RAG API | `python -m rag_system.api_server` or `python -m rag_system.main api --port 8001` | 8001 | `socketserver.TCPServer` (**requests are serialized**) | +| Ollama | `ollama serve` | 11434 | external | + +`python run_system.py` starts all four and aggregates their logs. + +The backend port is the module constant `PORT = 8000` in `backend/server.py` and the RAG API port defaults to `8001` in `start_server()`; neither is read from an environment variable. What *is* configurable is where each side looks for the others โ€” see [ยง6](#6-configuration-entry-points). + +### The browser talks to two origins + +The chat UI defaults to streaming (`enableStream` is `true` in `src/components/ui/session-chat.tsx`), and streaming goes **directly** from the browser to `:8001/chat/stream`, bypassing the gateway. Session CRUD, uploads, index management and the non-streaming chat fallback go to `:8000`. Any deployment must therefore expose **both** ports to the browser, and the frontend build must know both URLs (`NEXT_PUBLIC_API_URL`, `NEXT_PUBLIC_RAG_API_URL`). Both servers send `Access-Control-Allow-Origin: *`. + +--- + +## 2. Request Paths + +### 2.1 Streaming chat (the default UI path) + +```mermaid +sequenceDiagram + participant B as Browser + participant R as RAG API :8001 + participant O as Ollama :11434 + B->>R: POST /chat/stream {query, session_id, table_name, options} + R->>R: Agent.run(...) โ€” triage โ†’ retrieve โ†’ rerank โ†’ prune โ†’ synthesize + R->>O: enrichment model (routing, decomposition, verification) + R->>O: generation model (answer, streamed) + R-->>B: SSE: analyze, decomposition, retrieval_*, rerank_*, token, complete +``` + +`Agent.run()` executes the whole pipeline and the handler forwards each phase as an SSE event. See [ยง2.4](#24-in-process-pipeline-inside-the-rag-api). + +### 2.2 Non-streaming chat + +```mermaid +sequenceDiagram + participant B as Browser + participant G as Backend :8000 + participant R as RAG API :8001 + participant O as Ollama :11434 + B->>G: POST /sessions/{id}/messages {message, options} + G->>G: persist user message (SQLite) + G->>G: should_use_rag() โ€” deterministic gate, no model call + alt route = RAG + G->>R: POST /chat {query, session_id, table_name, options} + R-->>G: {answer, source_documents} + else route = direct + G->>O: generation model, thinking disabled end + G->>G: persist assistant message (SQLite) + G-->>B: {response, session, source_documents, used_rag} +``` + +**The gateway gate.** `should_use_rag(message, idx_ids, force_rag)` in `backend/server.py` is a pure function: `force_rag` โ†’ RAG; no linked indexes โ†’ direct; a whole-message smalltalk/assistant-meta allowlist regex (`hello`, `thanks!`, `who are you`, capped at six words) โ†’ direct; everything else โ†’ RAG. No LLM call, no file reads, no network. Measured at ~0.002 ms per message against ~750 ms for the enrichment-model router it replaced (Phase 2.3). + +It is deliberately biased toward RAG. The agent-side triage in [ยง2.4](#24-in-process-pipeline-inside-the-rag-api) still runs on every RAG request and can still return a direct answer, so the gate never has to decide correctly on its own โ€” it only has to avoid sending "hi" through a retrieval pipeline. `used_rag` in the response reports which way the gate went. + +### 2.3 Indexing - %% data edges - IDX -->|chunks & embeddings| LDB - RET -->|vector search| LDB - LOOP -->|LLM calls| OLLAMA - RP -->|LLM calls| OLLAMA - VER -->|LLM calls| OLLAMA - RP -->|rerank| HF +Uploads land in `shared_uploads/` (backend). Indexing is then triggered either per session (`POST :8000/sessions/{id}/index`) or per named index (`POST :8000/indexes/{id}/build`); both forward to `POST :8001/index`, which runs `IndexingPipeline.run(file_paths)` in the RAG API process. - BE -->|CRUD| SQL +```mermaid +flowchart LR + UP["shared_uploads/*"] --> DC["DocumentConverter
(Docling, OCR when available)"] + DC --> CH["DoclingChunker
(token-budgeted)"] + CH --> OV["OverviewBuilder
index_store/overviews/<id>.jsonl"] + CH --> EN["ContextualEnricher
(optional, enrichment model)"] + EN --> EM["Embeddings
harrier-oss-v1-0.6b"] + EM --> LT["LanceDB table
+ native FTS index"] + CH --> LC["Late-chunk encoder
(optional) โ†’ <table>_lc"] +``` + +Contextual enrichment returns copies of the chunks, so the late-chunk leg encodes the **original** chunk text while the main table stores the enriched text (with the original preserved in `metadata.original_text`). + +### 2.4 In-process pipeline inside the RAG API + +```mermaid +flowchart TD + Q["Query"] --> T{"Triage"} + T -->|direct_answer| DA["Generation model, streamed"] + T -->|rag_query| DEC{"Query decomposition
(enabled in 'default')"} + DEC -->|n sub-queries| PAR["Parallel per-sub-query retrieval (โ‰ค3 threads),
candidates pooled + deduped (arm H default)"] + DEC -->|single query| RP["RetrievalPipeline.run"] + PAR --> RR + RP --> RET["MultiVectorRetriever
FTS + vector, fused with RRF"] + RET --> LCM["Late-chunk retrieval + ยฑ1 merge (optional)"] + LCM --> RR["Reranker (on by default)
Qwen3-Reranker-4B"] + RR --> CE["Context expansion (ยฑwindow chunks)"] + CE --> PR["Provence sentence pruning (opt-in)"] + PR --> SY["Answer synthesis (generation model, streamed)"] + SY --> VF["Verifier โ†’ appends [Confidence: N%]"] ``` +With the shipped default (arm H, 2026-08-15) the decomposed path retrieves per sub-query, pools and de-duplicates the candidates, then runs ONE rerank and ONE synthesis over the union context. Sending `compose_sub_answers: true` keeps the older path: answer each sub-query, then compose the final answer from the sub-answers. + +Triage order in `Agent._triage_query_async`: the overview router runs first (an LLM call on the enrichment model, grounded in the document overviews loaded for the session); if there is conversation history the query short-circuits to `rag_query`; otherwise a fallback LLM triage picks `rag_query` / `direct_answer`. `force_rag=true` on the RAG API skips triage entirely. + +There is no GraphRAG path. `GraphExtractor`, `GraphRetriever`, `GraphQueryTranslator`, the `graph_query` triage outcome and the `retrieval.graph` / `graph_strategy` config blocks were **deleted on 2026-08-09** (roadmap item 2.5): the path was unreachable, GraphRAG loses on single-hop retrieval, its multi-hop gains are contested, and it costs 41โ€“57ร— at indexing and up to ~377ร— in query tokens ([`research/academic-evidence-2026.md`](research/academic-evidence-2026.md) ยง6). + --- -### Data-flow Narrative -1. **User** interacts with the Next.js UI; messages are posted via `src/lib/api.ts`. -2. **backend/server.py** receives JSON over HTTP, applies CORS, and proxies the request into `rag_system`. -3. **Agent Loop** decides (via _Triage_) whether to perform Retrieval-Augmented Generation (RAG) or direct LLM answering. -4. If RAG is chosen: - 1. **Retrieval Pipeline** fetches candidates from **LanceDB** using BM25 + dense vectors. - 2. **AI Reranker** (HF model) sorts snippets. - 3. **Answer Synthesiser** calls **Ollama** to write the final answer. -5. Answers can be **Verified** for grounding (optional flag). -6. Index-building is an offline path triggered from the UI โ€” PDF/๐Ÿ“„ files are chunked, embedded and stored in LanceDB. +## 3. Where State Lives + +| Store | Location | Written by | Contents | +|-------|----------|-----------|----------| +| SQLite | `backend/chat_data.db` (`DB_PATH`) | backend (all tables), RAG API (index metadata only) | `sessions`, `messages`, `session_documents`, `indexes`, `index_documents`, `session_indexes` | +| LanceDB | `./lancedb` (`storage.lancedb_uri`, `LANCEDB_PATH`) | RAG API | `text_pages_v4` (default table), `text_pages_` (per named index), `
_lc` (late-chunk vectors) | +| Overviews | `index_store/overviews/.jsonl`, plus `index_store/overviews/overviews.jsonl` as a global fallback | RAG API (indexing) | one-paragraph document summaries used by the agent's triage router (the gateway gate no longer reads them) | +| Uploads | `shared_uploads/` | backend | original uploaded files, prefixed with a UUID | + +**Chat messages are written by `backend/server.py` only** (`handle_session_chat`). The RAG API holds a `ChatDatabase` handle but never touches the `messages` or `sessions` tables. --- -## 2. Component Documents -The table below links to deep-dives for each major component. +## 4. Threading and Shared State -| **Component** | **Documentation** | -|---------------|-------------------| -| Agent Loop | [`system_overview.md`](system_overview.md) | -| Indexing Pipeline | [`indexing_pipeline.md`](indexing_pipeline.md) | -| Retrieval Pipeline | [`retrieval_pipeline.md`](retrieval_pipeline.md) | -| Verifier | [`verifier.md`](verifier.md) | -| Triage System | [`triage_system.md`](triage_system.md) | +* The **backend** is threaded (`ThreadingTCPServer`, `daemon_threads = True`). `backend/database.py` opens a fresh SQLite connection per call, so concurrent handlers are safe. +* The **RAG API is single-threaded**. One `/chat`, `/chat/stream` or `/index` request is served at a time; everything else queues. This is deliberate: the process holds one `RAG_AGENT` singleton whose pipeline config is mutated by per-request options. +* Because that config is shared, per-request retrieval options are applied to it โ€” but scoped to the request: the agent snapshots the pipeline config before applying overrides and restores it afterwards, so options no longer leak into later requests. The per-request generation `model` goes through a context manager (`_generation_model_override`) that restores the previous value and rejects model ids that do not match the active `LLM_BACKEND`. +* `factory.get_pipeline_config()` returns a `copy.deepcopy` of the profile, so the module-level `PIPELINE_CONFIGS` in `rag_system/main.py` is never mutated. --- -> **Change-management**: whenever architecture changes (new micro-service, different DB, etc.) update this overview diagram first, then individual component docs. \ No newline at end of file +## 5. Known Architectural Limitations + +These are real, current behaviours โ€” not planned work. See [`improvement_plan.md`](improvement_plan.md) for the fixes on the roadmap. + +1. **Streamed turns are persisted by a client callback, not by the stream itself.** `POST :8001/chat/stream` writes nothing to SQLite; when the stream completes, the browser posts the finished turn to `POST :8000/sessions/{id}/messages/save`, which stores both messages and derives the session title. A client that consumes the stream directly and skips that call gets no history. +2. **The RAG API serializes requests.** Concurrency is one in-flight RAG request per process. +3. **`enable_docling_chunk` defaults to `true` over HTTP.** `rag_system/api_server.py` maps the flag to `chunker_mode` (`"docling"` when true, `"legacy"` when false), so the Docling chunker runs unless the client sends `false` to select the legacy chunker. (`create_index_script.py` sets `"legacy"` explicitly.) +4. **Service ports are not configurable by environment variable** โ€” only the URLs used to reach them are. + +--- + +## 6. Configuration Entry Points + +| Concern | Where | +|---------|-------| +| Model defaults, pipeline profiles | `rag_system/main.py` (`OLLAMA_CONFIG`, `WATSONX_CONFIG`, `EXTERNAL_MODELS`, `PIPELINE_CONFIGS`) | +| Agent / pipeline construction | `rag_system/factory.py` (`get_agent`, `get_indexing_pipeline`, `get_pipeline_config`) | +| Active profile for the RAG API | `RAG_CONFIG_MODE` (default `default`; an unknown value silently falls back to `default`) | +| Service URLs and model overrides | environment variables โ€” see [`.env.example`](../.env.example) and [`system_overview.md`](system_overview.md#7-configuration) | + +--- + +## 7. Component Documents + +| Component | Documentation | +|-----------|---------------| +| Whole system, models, configuration | [`system_overview.md`](system_overview.md) | +| Why each component is built this way (evidence + eval numbers; "deliberately not implemented") | [`design_rationale.md`](design_rationale.md) | +| HTTP APIs (backend + RAG API) | [`api_reference.md`](api_reference.md) | +| Indexing pipeline | [`indexing_pipeline.md`](indexing_pipeline.md) | +| Retrieval pipeline | [`retrieval_pipeline.md`](retrieval_pipeline.md) | +| Verifier | [`verifier.md`](verifier.md) | +| Triage / routing | [`triage_system.md`](triage_system.md) | +| Roadmap | [`improvement_plan.md`](improvement_plan.md) | + +> **Change management**: when the topology changes (new process, new store, a new browser-facing origin), update this overview first, then the component docs. diff --git a/Documentation/deployment_guide.md b/Documentation/deployment_guide.md index 37a54e14..f3d234b3 100644 --- a/Documentation/deployment_guide.md +++ b/Documentation/deployment_guide.md @@ -1,22 +1,22 @@ -# ๐Ÿš€ RAG System Deployment Guide +# ๐Ÿš€ LocalGPT Deployment Guide -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ -This guide provides comprehensive instructions for deploying the RAG system using both Docker and direct development approaches. +This guide covers deploying LocalGPT with Docker or as directly-run processes. --- ## ๐ŸŽฏ Deployment Options -### Option 1: Docker Deployment (Production) ๐Ÿณ -- **Best for**: Production environments, containerized deployments, scaling -- **Pros**: Isolated, reproducible, easy to manage -- **Cons**: Slightly more complex setup, resource overhead +### Option 1: Docker Deployment ๐Ÿณ +- **Best for**: reproducible environments, keeping dependencies isolated +- **Pros**: one command, pinned base images, health-gated startup order +- **Cons**: slower first build, GPU passthrough is extra work -### Option 2: Direct Development (Development) ๐Ÿ’ป -- **Best for**: Development, debugging, customization -- **Pros**: Direct access to code, faster iteration, easier debugging -- **Cons**: More dependencies to manage +### Option 2: Direct Processes ๐Ÿ’ป +- **Best for**: development, debugging, GPU/MPS access +- **Pros**: direct access to code, faster iteration, native acceleration +- **Cons**: more dependencies to manage on the host --- @@ -32,13 +32,12 @@ This guide provides comprehensive instructions for deploying the RAG system usin #### **Recommended Requirements** - **CPU**: 8+ cores, 3.0GHz+ -- **RAM**: 32GB+ (for large models) +- **RAM**: 32GB+ - **Storage**: 200GB+ SSD -- **GPU**: NVIDIA GPU with 8GB+ VRAM (optional, for acceleration) +- **GPU**: NVIDIA GPU with 8GB+ VRAM (optional; Apple Silicon uses MPS) ### 1.2 Common Dependencies -**Both deployment methods require:** ```bash # Ollama (required for both approaches) curl -fsSL https://ollama.ai/install.sh | sh @@ -49,21 +48,17 @@ git 2.30+ ### 1.3 Docker-Specific Dependencies -**For Docker deployment:** ```bash -# Docker & Docker Compose Docker Engine 24.0+ -Docker Compose 2.20+ +Docker Compose plugin 2.20+ ``` -### 1.4 Direct Development Dependencies +### 1.4 Direct Deployment Dependencies -**For direct development:** ```bash -# Python & Node.js -Python 3.8+ -Node.js 16+ -npm 8+ +Python 3.10+ # 3.11 recommended; the images use python:3.11-slim +Node.js 20+ +npm 10+ ``` --- @@ -76,307 +71,384 @@ npm 8+ **Ubuntu/Debian:** ```bash -# Install Docker curl -fsSL https://get.docker.com -o get-docker.sh sudo sh get-docker.sh sudo usermod -aG docker $USER newgrp docker -# Install Docker Compose V2 sudo apt-get update sudo apt-get install docker-compose-plugin ``` **macOS:** ```bash -# Install Docker Desktop brew install --cask docker # Or download from: https://www.docker.com/products/docker-desktop ``` **Windows:** ```bash -# Install Docker Desktop with WSL2 backend +# Install Docker Desktop with the WSL2 backend # Download from: https://www.docker.com/products/docker-desktop ``` #### **Step 2: Clone Repository** ```bash -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT ``` #### **Step 3: Install Ollama** ```bash -# Install Ollama (runs locally even with Docker) +# Runs on the host by default, even with Docker curl -fsSL https://ollama.ai/install.sh | sh -# Start Ollama ollama serve -# In another terminal, install models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +# In another terminal +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` -#### **Step 4: Launch Docker System** +#### **Step 4: Launch** ```bash -# Start all containers using the convenience script +# Convenience script (local Ollama) ./start-docker.sh -# Or manually: +# Containerized Ollama instead +./start-docker.sh container + +# Or manually docker compose --env-file docker.env up --build -d ``` #### **Step 5: Verify Deployment** ```bash -# Check container status docker compose ps -# Test all endpoints -curl http://localhost:3000 # Frontend -curl http://localhost:8000/health # Backend -curl http://localhost:8001/models # RAG API +curl http://localhost:3000 # Frontend +curl http://localhost:8000/health # Backend +curl http://localhost:8001/health # RAG API curl http://localhost:11434/api/tags # Ollama ``` -### 2.2 Docker Management +### 2.2 Startup Order -#### **Container Operations** -```bash -# Start system -./start-docker.sh - -# Stop system -./start-docker.sh stop +`docker-compose.yml` gates the stack on health checks: -# View logs -./start-docker.sh logs +``` +rag-api (healthy: GET /health) -> backend (healthy: GET /health) -> frontend +``` -# Check status -./start-docker.sh status +`rag-api` loads the embedding model before it answers `/health` (the reranker +loads lazily, on the first reranked query), so +its check uses a 120s start period. Until it passes, `backend` stays in `created` +and `frontend` after it. This is expected on a cold start, not a hang โ€” watch +`docker compose logs -f rag-api`. -# Manual Docker Compose commands -docker compose ps # Check status -docker compose logs -f # Follow logs -docker compose down # Stop all containers -docker compose up --build -d # Rebuild and restart -``` +### 2.3 Docker Management -#### **Individual Container Management** ```bash -# Restart specific service -docker compose restart rag-api +# Convenience script +./start-docker.sh # start (local Ollama) +./start-docker.sh container # start (containerized Ollama) +./start-docker.sh stop # stop +./start-docker.sh logs # follow logs +./start-docker.sh status # container status +./start-docker.sh help # usage -# View specific service logs -docker compose logs -f backend +# Compose directly +docker compose ps +docker compose logs -f +docker compose down +docker compose --env-file docker.env up --build -d -# Execute commands in container -docker compose exec rag-api python -c "print('Hello')" +# One service +docker compose restart rag-api +docker compose logs -f backend +docker compose exec rag-api python -c "print('hello')" ``` +### 2.4 Compose Files + +| File | Contents | +|------|----------| +| `docker-compose.yml` | The full stack: `rag-api`, `backend`, `frontend`, and an optional `ollama` service behind the `with-ollama` profile | + +`docker-compose.yml` is the only compose file: it defaults to host Ollama and +adds the `ollama` container when you pass `--profile with-ollama`. + --- -## 3. ๐Ÿ’ป Direct Development +## 3. ๐Ÿ’ป Direct Deployment ### 3.1 Installation -#### **Step 1: Install Dependencies** - **Python Dependencies:** ```bash -# Clone repository -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT -# Create virtual environment (recommended) python -m venv venv source venv/bin/activate # On Windows: venv\Scripts\activate -# Install Python packages pip install -r requirements.txt ``` **Node.js Dependencies:** ```bash -# Install Node.js dependencies npm install ``` -#### **Step 2: Install and Configure Ollama** +### 3.2 Install and Configure Ollama ```bash -# Install Ollama curl -fsSL https://ollama.ai/install.sh | sh - -# Start Ollama ollama serve -# In another terminal, install models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +# In another terminal +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` -#### **Step 3: Launch System** +### 3.3 Launch **Option A: Integrated Launcher (Recommended)** ```bash -# Start all components with one command +# Development python run_system.py + +# Production: runs `npm run build`, then `next start` +python run_system.py --mode prod ``` -**Option B: Manual Component Startup** +The launcher records the launcher PID plus each child PID in +`logs/run_system.pid`, aggregates every service's stdout into `logs/.log`, +and restarts a required service that exits unexpectedly (checked every 30s). + +**Option B: Manual Component Startup โ€” all from the repository root** ```bash # Terminal 1: RAG API python -m rag_system.api_server # Terminal 2: Backend -cd backend && python server.py +python backend/server.py # Terminal 3: Frontend -npm run dev +npm run build && npm run start # or: npm run dev # Access at http://localhost:3000 ``` -#### **Step 4: Verify Installation** -```bash -# Check system health -python system_health_check.py - -# Test endpoints -curl http://localhost:3000 # Frontend -curl http://localhost:8000/health # Backend -curl http://localhost:8001/models # RAG API -``` +> Relative paths (`backend/chat_data.db`, `lancedb/`, `index_store/`, +> `shared_uploads/`) resolve against the working directory. Always start from the +> repository root. -### 3.2 Direct Development Management +### 3.4 Direct Deployment Management -#### **System Operations** ```bash -# Start system -python run_system.py - -# Check system health -python system_health_check.py - -# Stop system -# Press Ctrl+C in terminal running run_system.py +python run_system.py --health # HTTP checks; exit 1 if a required service fails +python run_system.py --logs-only # tail logs/*.log from another shell +python run_system.py --stop # terminate everything in logs/run_system.pid +python run_system.py --no-frontend # Ollama + RAG API + backend only +python system_health_check.py # deep check incl. a real embedding + query ``` -#### **Individual Component Management** -```bash -# Start components individually -python -m rag_system.api_server # RAG API on port 8001 -cd backend && python server.py # Backend on port 8000 -npm run dev # Frontend on port 3000 - -# Development tools -npm run build # Build frontend for production -pip install -r requirements.txt --upgrade # Update Python packages -``` +`--stop` reads `logs/run_system.pid`, kills the launcher first (so its monitor +cannot restart anything), then each service and its descendants with SIGTERM, +escalating to SIGKILL after 10s. It exits non-zero if there is no pidfile. --- -## 4. Architecture Comparison +## 4. Architecture ### 4.1 Docker Architecture ```mermaid graph TB subgraph "Docker Containers" - Frontend[Frontend Container
Next.js
Port 3000] - Backend[Backend Container
Python API
Port 8000] - RAG[RAG API Container
Document Processing
Port 8001] + Frontend[frontend
Next.js
Port 3000] + Backend[backend
Gateway + SQLite
Port 8000] + RAG[rag-api
Indexing + Retrieval
Port 8001] end - - subgraph "Local System" + + subgraph "Host" Ollama[Ollama Server
Port 11434] end - + + Browser --> Frontend + Browser -. "SSE /chat/stream" .-> RAG Frontend --> Backend Backend --> RAG RAG --> Ollama + Backend --> Ollama ``` -### 4.2 Direct Development Architecture +`backend` and `rag-api` both mount `./backend`, `./lancedb` and `./index_store`, +and both point `DB_PATH` at `/app/backend/chat_data.db`, so they share one SQLite +file and one vector store. + +### 4.2 Direct Architecture ```mermaid graph TB subgraph "Local Processes" - Frontend[Next.js Dev Server
Port 3000] - Backend[Python Backend
Port 8000] - RAG[RAG API
Port 8001] + Frontend[Next.js
Port 3000] + Backend[backend/server.py
Port 8000] + RAG[rag_system.api_server
Port 8001] Ollama[Ollama Server
Port 11434] end - + + Browser --> Frontend + Browser -. "SSE /chat/stream" .-> RAG Frontend --> Backend Backend --> RAG RAG --> Ollama + Backend --> Ollama ``` +### 4.3 Concurrency + +- `backend/server.py` uses a `ThreadingTCPServer`, so it handles requests in + parallel and stays responsive during a long RAG call. +- `rag_system/api_server.py` uses a plain single-threaded `TCPServer`. **RAG + requests are serialised** โ€” one chat or indexing run at a time. Plan capacity + for a single concurrent user of the RAG API, or put a queue in front of it. +- The backend allows `RAG_API_TIMEOUT` (default 600s) for a chat call and + `RAG_API_INDEX_TIMEOUT` (default 3600s) for indexing, returning 504 on timeout + and 502 when the RAG API is unreachable โ€” both with a JSON error body. + --- ## 5. Configuration ### 5.1 Environment Variables +Every variable below is read by code; the value shown is the default when unset. +`.env.example` carries the same list. + +| Variable | Default | Read by | +|----------|---------|---------| +| `OLLAMA_HOST` | `http://localhost:11434` | `rag_system/main.py`, `backend/ollama_client.py` | +| `RAG_API_URL` | `http://localhost:8001` | `backend/server.py` | +| `NEXT_PUBLIC_API_URL` | `http://localhost:8000` | `src/lib/api.ts` (inlined at build time) | +| `NEXT_PUBLIC_RAG_API_URL` | `http://localhost:8001` | `src/lib/api.ts` (inlined at build time) | +| `DB_PATH` | `backend/chat_data.db` | `backend/database.py` | +| `LANCEDB_PATH` | `./lancedb` | `rag_system/main.py` (pipeline profiles), `backend/database.py`, `system_health_check.py` | +| `GENERATION_MODEL` | `qwen3.5:9b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | same | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | `rag_system/main.py` | +| `RERANKER_MODEL` | `Qwen/Qwen3-Reranker-4B` (loaded lazily on the first reranked query) | `rag_system/main.py` | +| `RAG_CONFIG_MODE` | `default` | `rag_system/api_server.py` | +| `RAG_API_TIMEOUT` | `600` | `backend/server.py` | +| `RAG_API_INDEX_TIMEOUT` | `3600` | `backend/server.py` | +| `LLM_BACKEND` | `ollama` | `rag_system/main.py` | +| `HF_TOKEN` | unset | HuggingFace downloads | + #### **Docker Configuration (`docker.env`)** ```bash -# Ollama Configuration OLLAMA_HOST=http://host.docker.internal:11434 - -# Service Configuration NODE_ENV=production RAG_API_URL=http://rag-api:8001 NEXT_PUBLIC_API_URL=http://localhost:8000 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 +DB_PATH=/app/backend/chat_data.db +LANCEDB_PATH=/app/lancedb +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B ``` -#### **Direct Development Configuration** +All compose services declare +`extra_hosts: ["host.docker.internal:host-gateway"]`, so `host.docker.internal` +resolves on Linux as well as macOS and Windows. + +`NEXT_PUBLIC_*` are **build-time** values for Next.js. `docker-compose.yml` passes +them as build args to `Dockerfile.frontend`; setting them only at runtime has no +effect on an already-built image. If the browser must reach the services under a +different hostname, set them and rebuild: + +```bash +NEXT_PUBLIC_API_URL=https://gpt.example.com/api \ +NEXT_PUBLIC_RAG_API_URL=https://gpt.example.com/rag \ +docker compose --env-file docker.env up --build -d frontend +``` + +#### **Direct Deployment Configuration** ```bash -# Environment variables are set automatically by run_system.py -# Override in environment if needed: +# run_system.py inherits your shell environment unchanged and adds only +# NODE_ENV=production to the Python services in --mode prod. export OLLAMA_HOST=http://localhost:11434 export RAG_API_URL=http://localhost:8001 +python run_system.py ``` ### 5.2 Model Configuration -#### **Default Models** -```python -# Embedding Models -EMBEDDING_MODELS = [ - "Qwen/Qwen3-Embedding-0.6B", # Fast, 1024 dimensions - "Qwen/Qwen3-Embedding-4B", # High quality, 2048 dimensions -] +Defaults live in `rag_system/main.py` and are overridable by environment variable. -# Generation Models -GENERATION_MODELS = [ - "qwen3:0.6b", # Fast responses - "qwen3:8b", # High quality -] -``` +| Role | Default | Documented options | +|------|---------|--------------------| +| Generation | `qwen3.5:9b` | `qwen3.6:27b` (high-end, ~17GB), `qwen3.5:4b` (light) | +| Enrichment / utility | `qwen3.5:4b` | `qwen3.5:2b` (light) | +| Embedding | `microsoft/harrier-oss-v1-0.6b` (MIT, 1024 dims, ~1.2 GB) | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context โ€” multilingual / long-context corpora), `Qwen/Qwen3-Embedding-0.6B` (1024 dims) | +| Reranker (**on by default**) | `Qwen/Qwen3-Reranker-4B` | `BAAI/bge-reranker-v2-m3` (low latency), `answerdotai/answerai-colbert-small-v1`, `Qwen/Qwen3-Reranker-0.6B` | -### 5.3 Performance Tuning - -#### **Memory Settings** -```bash -# For Docker: Increase memory allocation -# Docker Desktop โ†’ Settings โ†’ Resources โ†’ Memory โ†’ 16GB+ +`GET /models` on either service reports what is actually selectable: +Ollama tags are split into generation and embedding lists by a substring match +(`embed`, `bge`, `embedding` on the RAG API; the backend also matches `text`), and +the HuggingFace embedding models are appended. A tag whose name happens to contain +one of those substrings will be classified as an embedding model. -# For Direct Development: Monitor with -htop # or top on macOS -``` +**Changing the embedding model requires re-indexing.** Vector width is measured +from the loaded model; writing a different width into an existing LanceDB table +fails with an explicit error. -#### **Model Settings** -```python -# Batch sizes (adjust based on available RAM) -EMBEDDING_BATCH_SIZE = 50 # Reduce if OOM -ENRICHMENT_BATCH_SIZE = 25 # Reduce if OOM +### 5.3 Performance Tuning -# Chunk settings -CHUNK_SIZE = 512 # Text chunk size -CHUNK_OVERLAP = 64 # Overlap between chunks -``` +There are no `SEARCH_CONFIG` / `CHUNK_OVERLAP` globals. The knobs are pipeline +config keys and per-request fields. + +**Indexing throughput** โ€” `PIPELINE_CONFIGS[]["indexing"]` in +`rag_system/main.py`, or `batch_size_embed` / `batch_size_enrich` on +`POST /index`: + +| Key | `default` | `fast` | Request field | +|-----|-----------|--------|---------------| +| `embedding_batch_size` | 50 | 100 | `batch_size_embed` (default 50) | +| `enrichment_batch_size` | 10 | 50 | `batch_size_enrich` (default 25) | + +**Chunking** โ€” `chunk_size` on `POST /index` (default 512) feeds the Docling +chunker's `max_tokens`, or the legacy chunker's `max_chunk_size` with +`min_chunk_size = chunk_size // 4`. When no `chunking.chunk_size` is present at +all (for example the `python -m rag_system.main index` CLI path), the pipeline +falls back to 1500. There is no `chunk_overlap` setting. + +**Contextual enrichment** is the most expensive part of indexing: one LLM call per +chunk. Disable it with `enable_enrich: false` for the fastest ingest. + +**Query cost** โ€” the biggest lever is the profile: + +| | `default` | `fast` | +|---|---|---| +| `retrieval.search_type` | `hybrid` | `vector_only` | +| `retrieval_k` | 20 | 10 | +| `reranker.enabled` | true | false | +| `query_decomposition.enabled` | true | false | +| `verification.enabled` | true | false | +| `retrieval.latechunk.enabled` | true | false | + +Select the profile with `RAG_CONFIG_MODE=fast` for the RAG API, or `--mode fast` +for the CLI. Individual toggles (`ai_rerank`, `verify`, `query_decompose`, +`context_expand`, `retrieval_k`, `reranker_top_k`) can also be sent per request. + +**Memory** โ€” the embedding model stays resident in the RAG API process +(`microsoft/harrier-oss-v1-0.6b`, ~1.2 GB). Reranking is on by default, but the +reranker loads lazily: the first reranked query pulls ~7.5 GB of +`Qwen/Qwen3-Reranker-4B` weights alongside it โ€” see +[`../eval/DECISIONS.md`](../eval/DECISIONS.md) for the quality/latency trade that +decision rests on. Enabling late chunking loads a second copy of the embedding +model. --- @@ -384,77 +456,84 @@ CHUNK_OVERLAP = 64 # Overlap between chunks ### 6.1 System Monitoring -#### **Health Checks** ```bash -# Comprehensive system check +# Health curl -f http://localhost:3000 && echo "โœ… Frontend OK" curl -f http://localhost:8000/health && echo "โœ… Backend OK" -curl -f http://localhost:8001/models && echo "โœ… RAG API OK" +curl -f http://localhost:8001/health && echo "โœ… RAG API OK" curl -f http://localhost:11434/api/tags && echo "โœ… Ollama OK" + +# Or, for a direct deployment +python run_system.py --health ``` -#### **Performance Monitoring** -```bash -# Docker monitoring -docker stats +`GET /health` on the backend returns Ollama reachability, the model list and +database stats; on the RAG API it returns `{"status": "ok"}` as soon as the agent +has finished loading. -# Direct development monitoring -htop # Overall system -nvidia-smi # GPU usage (if available) +```bash +# Resource usage +docker stats # Docker +htop # host +nvidia-smi # GPU, if present ``` ### 6.2 Log Management #### **Docker Logs** ```bash -# All services docker compose logs -f - -# Specific service docker compose logs -f rag-api - -# Save logs to file docker compose logs > system.log 2>&1 ``` -#### **Direct Development Logs** +#### **Direct Deployment Logs** ```bash -# Logs are printed to terminal -# Redirect to file if needed: -python run_system.py > system.log 2>&1 +# run_system.py writes per-service files +tail -f logs/system.log logs/rag-api.log logs/backend.log logs/frontend.log + +# or use the launcher's own tailer from a second shell +python run_system.py --logs-only ``` ### 6.3 Backup and Restore +Everything persistent is a host directory, including under Docker (all mounts are +bind mounts; the only named volume is `ollama_data` for the optional Ollama +container). + #### **Data Backup** ```bash -# Create backup directory -mkdir -p backups/$(date +%Y%m%d) +# Stop first so SQLite and LanceDB are not mid-write +./start-docker.sh stop # Docker +# or: python run_system.py --stop -# Backup databases and indexes -cp -r backend/chat_data.db backups/$(date +%Y%m%d)/ -cp -r lancedb backups/$(date +%Y%m%d)/ -cp -r index_store backups/$(date +%Y%m%d)/ +mkdir -p backups/$(date +%Y%m%d) +cp backend/chat_data.db backups/$(date +%Y%m%d)/ +tar czf backups/$(date +%Y%m%d)/lancedb.tar.gz lancedb/ +tar czf backups/$(date +%Y%m%d)/index_store.tar.gz index_store/ +tar czf backups/$(date +%Y%m%d)/shared_uploads.tar.gz shared_uploads/ -# For Docker: also backup volumes -docker compose down -docker run --rm -v rag_system_old_ollama_data:/data -v $(pwd)/backups:/backup alpine tar czf /backup/ollama_models_$(date +%Y%m%d).tar.gz -C /data . +# Only if you use the containerized Ollama, back up its named volume too. +# The project name is the directory name, so the volume is localgpt_ollama_data. +docker run --rm -v localgpt_ollama_data:/data -v $(pwd)/backups:/backup \ + alpine tar czf /backup/ollama_models_$(date +%Y%m%d).tar.gz -C /data . ``` #### **Data Restore** ```bash -# Stop system -./start-docker.sh stop # Docker -# Or Ctrl+C for direct development +./start-docker.sh stop # or: python run_system.py --stop -# Restore files -cp -r backups/YYYYMMDD/* ./ +cp backups/YYYYMMDD/chat_data.db backend/ +tar xzf backups/YYYYMMDD/lancedb.tar.gz +tar xzf backups/YYYYMMDD/index_store.tar.gz -# Restart system -./start-docker.sh # Docker -python run_system.py # Direct development +./start-docker.sh # or: python run_system.py ``` +Restore the SQLite file and `lancedb/` together โ€” the database rows point at +LanceDB table names, so a mismatched pair leaves indexes that resolve to nothing. + --- ## 7. Troubleshooting @@ -463,109 +542,96 @@ python run_system.py # Direct development #### **Port Conflicts** ```bash -# Check what's using ports lsof -i :3000 -i :8000 -i :8001 -i :11434 -# For Docker: Stop conflicting containers +# Docker ./start-docker.sh stop -# For Direct: Kill processes -pkill -f "npm run dev" -pkill -f "server.py" -pkill -f "api_server" +# Direct +python run_system.py --stop ``` +If a required port (8001 or 8000) is already taken, `run_system.py` logs +`Port โ€ฆ already in use, skipping โ€ฆ` and aborts with "System startup failed". Port +11434 in use is treated as "Ollama already running" and reused; port 3000 in use is +tolerated because the frontend is optional. + #### **Docker Issues** ```bash -# Docker daemon not running -docker version # Check if daemon responds - -# Restart Docker Desktop (macOS/Windows) -# Or restart docker service (Linux) -sudo systemctl restart docker - -# Clear Docker cache -docker system prune -f +docker version # daemon reachable? +sudo systemctl restart docker # Linux +docker system prune -f # clear build cache ``` #### **Ollama Issues** ```bash -# Check Ollama status curl http://localhost:11434/api/tags -# Restart Ollama pkill ollama ollama serve -# Reinstall models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` +#### **Backend returns 502/504 for every chat** +The backend could not reach the RAG API (502) or the call timed out (504) โ€” both +come back as JSON error bodies. Check `RAG_API_URL` (must be +`http://rag-api:8001` inside Docker, not `localhost`) and that `rag-api` is +healthy. + ### 7.2 Performance Issues #### **Memory Problems** ```bash -# Check memory usage free -h # Linux vm_stat # macOS -docker stats # Docker containers +docker stats # containers -# Solutions: -# 1. Increase system RAM -# 2. Reduce batch sizes in configuration -# 3. Use smaller models (qwen3:0.6b instead of qwen3:8b) +# Options: +# 1. Switch to RAG_CONFIG_MODE=fast +# 2. Switch reranking off (ai_rerank=false) โ€” it is on by default and lazily +# loads ~7.5 GB of reranker weights on the first reranked query +# 3. Use a smaller generation model (GENERATION_MODEL=qwen3.5:4b) +# 4. Disable late chunking so a second embedding model is not loaded ``` #### **Slow Response Times** ```bash -# Check model loading -curl http://localhost:11434/api/tags - -# Monitor component response times -time curl http://localhost:8001/models +# Where is the time going? +docker compose logs -f rag-api # stage-by-stage output -# Solutions: -# 1. Use SSD storage -# 2. Increase CPU cores -# 3. Use GPU acceleration (if available) +time curl -s http://localhost:8001/health # process responsive? ``` +Remember requests to the RAG API are serialised โ€” a query that appears slow may be +queued behind an indexing run. + --- ## 8. Production Considerations ### 8.1 Security -#### **Network Security** -```bash -# Use reverse proxy (nginx/traefik) for production -# Enable HTTPS/TLS -# Restrict port access with firewall -``` +LocalGPT ships with **no authentication** and permissive CORS +(`Access-Control-Allow-Origin: *`) on both HTTP services. Before exposing it: -#### **Data Security** -```bash -# Enable authentication in production -# Encrypt sensitive data -# Regular security updates -``` +- Put a reverse proxy (nginx, Caddy, Traefik) in front and terminate TLS there +- Add authentication at the proxy +- Publish only port 3000; keep 8000 and 8001 on an internal network +- Note the browser calls the RAG API directly for streaming, so port 8001 must be + reachable by clients (or proxied) if you leave streaming enabled ### 8.2 Scaling -#### **Horizontal Scaling** -```bash -# Use Docker Swarm or Kubernetes -# Load balance frontend and backend -# Scale RAG API instances based on load -``` +`docker compose up --scale` does **not** work with the shipped compose file: every +service sets a fixed `container_name` and publishes a fixed host port. Scaling +requires removing those first. -#### **Resource Optimization** -```bash -# Use dedicated GPU nodes for AI workloads -# Implement model caching -# Optimize batch processing -``` +Beyond that, the RAG API is single-threaded and holds mutable state (the resident +agent, its per-session in-memory chat history and the semantic cache), so running +several replicas behind a load balancer needs sticky sessions at minimum. The +realistic path is a single RAG API with a queue in front of it. --- @@ -573,26 +639,24 @@ time curl http://localhost:8001/models ### 9.1 Deployment Verification -Your deployment is successful when: - -- โœ… All health checks pass +- โœ… `docker compose ps` shows all services healthy (or `python run_system.py --health` exits 0) - โœ… Frontend loads at http://localhost:3000 -- โœ… You can create document indexes +- โœ… You can create a document index - โœ… You can chat with uploaded documents -- โœ… No error messages in logs +- โœ… No errors in `docker compose logs` / `logs/` -### 9.2 Performance Benchmarks +### 9.2 What to Expect -**Acceptable Performance:** -- Index creation: < 2 minutes per 100MB document -- Query response: < 30 seconds for complex questions -- Memory usage: < 8GB total system memory +Throughput depends on hardware and model size, so treat these as shape rather than +guarantees: -**Optimal Performance:** -- Index creation: < 1 minute per 100MB document -- Query response: < 10 seconds for complex questions -- Memory usage: < 16GB total system memory +- Cold start is dominated by model downloads and the RAG API loading the embedding + model (the reranker loads lazily, on the first reranked query) +- Indexing is dominated by contextual enrichment (one LLM call per chunk) +- Query latency is dominated by generation; `fast` mode removes reranking, + decomposition and verification +- Concurrency is one RAG request at a time --- -**Happy Deploying! ๐Ÿš€** \ No newline at end of file +**Happy Deploying! ๐Ÿš€** diff --git a/Documentation/design_rationale.md b/Documentation/design_rationale.md new file mode 100644 index 00000000..e94eee1e --- /dev/null +++ b/Documentation/design_rationale.md @@ -0,0 +1,752 @@ +# Design Rationale + +_Revision: 2026-08-09. Roadmap item [3.1](research_roadmap.md#phase-3--documentation-make-the-evidence-part-of-the-repos-argument)._ + +This document answers one question per component: **why does localGPT do it this +way?** It describes only what ships in the current tree. Anything planned, +proposed or merely promising lives in [`research_roadmap.md`](research_roadmap.md) +and [`improvement_plan.md`](improvement_plan.md), not here. + +## The method + +Three artefacts, in a fixed order: + +1. **Evidence** โ€” [`research/`](research/) holds three August-2026 sweeps of + primary sources (papers, model cards, first-party engineering blogs), each + claim graded *established* / *emerging* / *contested* and each carrying its + own "could not verify" appendix. These documents describe the field, **not + this repo**. +2. **Our own eval** โ€” [`eval/`](../eval/) holds a 72-query gold set over three + corpora, a recall/nDCG runner, a binary groundedness judge validated against + hand labels, and an end-to-end smoke test. See [`eval/README.md`](../eval/README.md). +3. **A decision** โ€” nothing changes a default without a measured delta on (2). + The decisions and their numbers are in [`eval/DECISIONS.md`](../eval/DECISIONS.md) + and [`eval/decisions/`](../eval/decisions/). + +The order matters because **it repeatedly produced the opposite answer from the +literature.** Three times in one week: + +| The evidence said | Our eval measured | What shipped | +|---|---|---| +| "Cross-encoder rerank: **YES, unconditionally**, +17.2 pp MRR@3" ([`component-map-2026.md` ยง5.1](research/component-map-2026.md), ยง6.5) | `bge-reranker-v2-m3` on top of the new first stage: **โˆ’0.022 nDCG@10** on `mixed`, **โˆ’0.058** on `docs`, for ~1.6 s/query | Reranking initially **off** by default ([`DECISIONS.md` ยง1](../eval/DECISIONS.md)); turned **on** with score-threshold selection in arm G, 2026-08-14 | +| Decomposition helps when applied at *reranking* rather than first-stage (2026 MultiConIR/SSRB, [`component-map-2026.md` ยง6.2](research/component-map-2026.md)) | On the 6 queries that genuinely decompose: **โˆ’0.046** (`max`) / **โˆ’0.012** (`mean`) nDCG@10 | Shape change shipped; sub-query scoring at rerank now applies whenever the reranker is on (arm G), and the pooled first stage is the default decomposition path (arm H, 2026-08-15) ([`phase2-pipeline.md` ยง3](../eval/decisions/phase2-pipeline.md)) | +| Bigger instruction-tuned embedders lead the boards; the shipped default was the 8 GB `Qwen3-Embedding-4B` | A 1.2 GB MIT model **dominated** it: mixed nDCG@10 **0.915 vs 0.875**, ~3ร— lower latency, ~7ร— less memory | `microsoft/harrier-oss-v1-0.6b` as the default ([`embedder.md` gate section](../eval/decisions/embedder.md)) | + +That is the whole method: the literature nominates candidates, our gold set +decides. Where the two disagree, the gold set wins and the disagreement is +written down rather than smoothed over. + +Every number below is traceable to a file under `eval/`. Where a claim has no +number, it says so. + +--- + +## 1. Parsing / OCR + +**What ships.** Every document goes through [docling](https://github.com/docling-project/docling) +(`rag_system/ingestion/document_converter.py`). PDFs take one of two converters: +a text-layer probe (`_pdf_has_text`, PyMuPDF) decides whether OCR runs at all โ€” +if *any* page has extractable text, the whole document is converted without OCR. +DOCX / HTML / MD go through a third general converter; `.txt` is read directly +and fenced. Conversion returns `(markdown, metadata, DoclingDocument)` so the +chunker can use the element tree rather than re-parsing markdown. + +The OCR engine is **probed, not configured**: `build_ocr_options()` walks +`OcrMac โ†’ EasyOCR โ†’ RapidOCR โ†’ tesserocr โ†’ tesseract-cli` and picks the first +whose backend is actually importable on this host. There is no VLM parser. + +**Why.** The 2026 evidence ([`component-map-2026.md` ยง1.5](research/component-map-2026.md)) +says the local pipeline is "Docling as orchestration + a 0.9โ€“1.2B specialist VLM +as the parsing engine", and that traditional parsers "survive as the fast path, +not the quality path". We ship the orchestration half and not the VLM half, on +purpose: + +* The VLM spike ([`glm-ocr-spike.md`](../eval/decisions/glm-ocr-spike.md)) is + **GO-LATER**, not NO-GO. It demonstrated a real, large win on the one document + class that matters โ€” a degraded scanned invoice where GLM-OCR read **30/30 + table cells** and the current chain lost every price and 4 of 5 part numbers. +* Three defects block adoption: Ollama's Modelfile ignores prompts (so GLM-OCR's + table/formula modes are unreachable), some pages are deterministically + transcribed twice, and docling flattens the model's pipe tables to + `tables: 0`. None of them is code we should write blind. +* **There is no OCR eval.** `eval/corpora/` contains only digital-born PDFs with + clean text layers, so nothing in the harness exercises the OCR branch. Adopting + a parser on leaderboard position alone would violate the gate this repo runs on + โ€” and the spike showed exactly why: the roadmap's original "#1 on OmniDocBench, + beats GPT-5.2 by ~10 points" line did not survive source verification (GLM-OCR + is **third** on v1.6_full, behind PaddleOCR-VL-1.6 and MinerU2.5-Pro). + +The probe order was also the cheap win: the RapidOCR probe used to test for the +stale module name `rapidocr_onnxruntime`, so a host with `rapidocr` 3.x installed +silently fell all the way through to the Tesseract CLI. It now accepts either +name (`OCR_BACKENDS`, `document_converter.py`), and on this machine RapidOCR is +what resolves. + +**What would change the decision.** Build a scanned/tabular corpus with ground +truth under `eval/corpora/`, then A/B GLM-OCR against the *fixed* classic chain +**and** against docling's other 2026 presets (`lightonocr`, `dots_ocr`, +`nanonets_ocr2`) โ€” one column in a table, not a foregone winner. Independently of +any VLM: `pip install ocrmac` on macOS is a free upgrade over the current chain, +and the text-layer probe should become per-page (today a scanned insert inside a +digital PDF gets no OCR at all). + +## 2. Chunking + index-time enrichment + +**What ships.** `DoclingChunker` (`rag_system/ingestion/docling_chunker.py`) is +the default (`chunker_mode: "docling"`). It walks the `DoclingDocument` element +tree, emits **tables, code and figures as atomic chunks**, sentence-packs +paragraph nodes up to a token budget, and attaches the heading path and block +type to every chunk's metadata. Token counting uses the *embedding model's own* +tokenizer. Budget is 512 tokens for HTTP index builds +(`rag_system/api_server.py`) and falls back to 1500 for the CLI path, because no +profile sets `chunking.chunk_size`. Overlap is one sentence. + +**Index-time contextual enrichment is on** in the `default` profile +(`contextual_enricher: {enabled: True, window_size: 1}`, `rag_system/main.py`): +`ContextualEnricher` asks the enrichment model for a 2โ€“5 sentence situating +summary per chunk from its ยฑ1 neighbours and prepends it to the indexed text, +keeping the original text in metadata (`rag_system/indexing/contextualizer.py`). + +**Late chunking** is implemented (`rag_system/indexing/latechunk.py`: embed the +whole document, mean-pool inside chunk spans) and enabled in the `default` +profile, but `POST /index` defaults `enable_latechunk` to `false`, so HTTP builds +and CLI builds differ โ€” tracked as [`improvement_plan.md`](improvement_plan.md) +ยง3.6, not defended here. When a late-chunk table exists, its hits are merged with +their ยฑ1 siblings before reranking (`retrieval_pipeline.py::_first_stage`). + +**Why.** + +* **Boring chunking is the right default.** Every controlled study since 2024 + puts the *total* spread across all chunking methods at ~9 points of recall โ€” + Chroma's 2024 report and a 2026 eight-method / nine-dataset replication both + land on recursive/fixed splitting as the cost-effective winner, with LLM-driven + chunkers (DenseX 69.1 Acc@5, 15+ hours) far behind + ([`component-map-2026.md` ยง2.1, ยง2.5](research/component-map-2026.md)). The + same source's summary is blunt: "chunking is not where your quality is." +* **Structure-awareness is nearly free** when the parser already hands you an + element tree, which is why tables and code are atomic + ([`component-map-2026.md` ยง2.5](research/component-map-2026.md)). +* **Contextual enrichment is the one chunking-adjacent intervention with a large + measured effect** โ€” 35% top-20 failure-rate reduction from contextual + embeddings alone, 67% stacked with BM25 and reranking (Anthropic, via + [`component-map-2026.md` ยง2.3](research/component-map-2026.md)). It is an + offline cost, which is the only reason it is affordable here. +* **Late chunking is deliberately conditional.** The evidence grades it + *contested* โ€” efficient, but losing relevance to contextual retrieval, with the + sign flipping by corpus ([`component-map-2026.md` ยง2.2](research/component-map-2026.md)). + It is a fix for cross-reference breakage, not a general upgrade, which matches + shipping it as a per-build flag rather than a mandate. + +**What would change the decision.** Chunking has never been A/B'd on our gold +set โ€” no `eval/` number defends the 512/1500 budget or the one-sentence overlap. +If chunking is ever revisited, re-run it *after* any local-reader upgrade: the +same literature that says compression benefit shrinks as the reader gets stronger +says the identical thing about chunking ([`component-map-2026.md` ยง2.4](research/component-map-2026.md)). + +## 3. Embeddings + +**What ships.** `microsoft/harrier-oss-v1-0.6b` (MIT, 1024-dim, 1.2 GB) is the +default (`rag_system/main.py::EXTERNAL_MODELS`), overridable with +`EMBEDDING_MODEL`. Queries get the `Instruct: {task}\nQuery: {text}` prefix; +**documents do not** (`representations.py::default_query_instruction` + +`apply_query_instruction`, resolved per-pipeline by +`retrieval_pipeline.py::_query_instruction`, which honours +`config["embedding_instruction"]` then `EMBEDDING_INSTRUCTION` then the model +family's default). The prefix applies only to families that were trained with one +(`qwen3-embedding`, `harrier`); everything else gets `""`. + +Two index-format guarantees ship alongside it, both in +`rag_system/indexing/embedders.py` and `rag_system/retrieval/retrievers.py`: + +* **Per-table embedder identity.** Every table records the model that wrote it + plus a `normalized` flag in the Arrow schema metadata (with a JSON sidecar + fallback). Indexing into or querying a table written by a different embedder + raises `EmbedderMismatchError`. +* **L2 normalization** at write and query time, gated on that marker, so + LanceDB's default L2 ordering *is* the cosine ordering both model cards + specify. Legacy unmarked tables keep working unnormalized, with a warning. The + default table name moved to `text_pages_v4` so the shipped default starts clean. + +**Why.** + +* **The prefix**: every top-2026 embedder is instruction-tuned, all use the same + format, all specify no instruction on the document side, and Qwen3-Embedding + reports 1โ€“5% from the prefix alone + ([`component-map-2026.md` ยง3.5](research/component-map-2026.md)). The audit + found the repo was sending **no prefix at all** + ([`embedder.md` ยง1](../eval/decisions/embedder.md)). Our measurement is more + nuanced than the card: on harrier the prefix helps both metrics + (mixed nDCG@10 0.908 โ†’ 0.915), on Qwen3-0.6B it is a ranking-vs-recall trade + that washes out after a cross-encoder. Because only the query vector changes, + turning it on invalidated no index. +* **The model**: not chosen from a leaderboard. Measured head-to-head on our gold + set, harrier-0.6b beat the then-shipped 8 GB `Qwen3-Embedding-4B` on mixed + nDCG@10 (**0.915 vs 0.875**), docs nDCG@10 (**0.759 vs 0.638**), recall@5 and + recall@10, at ~3ร— lower latency and ~7ร— less memory + ([`embedder.md`, gate validation](../eval/decisions/embedder.md)). MIT is also + strictly more permissive than Apache-2.0. +* **The identity marker exists because the *previous* guard could not have caught + this swap.** It compared vector *width* only, and harrier-0.6b and + Qwen3-Embedding-0.6B are both 1024-dim โ€” exactly the swap this adoption makes + people likely to perform. It would have appended mutually unintelligible + vectors to a live table, silently ([`DECISIONS.md` ยง2](../eval/DECISIONS.md)). +* **Normalization was adopted for card-conformance, not for a number, and the + page says so.** Measured on `mixed`: nDCG@10 0.915 โ†’ **0.911**, recall@5 0.917 โ†’ + 0.931. A wash. It ships because the previous behaviour was neither cosine nor + intended, not because it won. + +**Post-adoption verification.** `mixed`, 317 chunks, 72 queries, first stage only: +recall@5 **0.944**, recall@10 0.958, recall@20 1.000, nDCG@10 **0.913** +([`DECISIONS.md` ยง4](../eval/DECISIONS.md)). Note that the `docs` and `mixed` +corpora index live `Documentation/*.md` โ€” **editing this file moves these +numbers.** Only compare runs made against the same tree. + +**What would change the decision.** `Qwen/Qwen3-Embedding-4B` stays a documented +option for multilingual or long-context (32K) corpora, which this English, +digital-born gold set does not exercise โ€” keep the prefix on for it, worth +0.059 +nDCG@10. Any embedder change forces a re-index **and** re-opens ยง6: the reranker's +value is a function of first-stage quality, and these two cannot be decided +independently. + +## 4. Hybrid retrieval + RRF + +**What ships.** `MultiVectorRetriever.retrieve` (`rag_system/retrieval/retrievers.py`) +runs LanceDB's native full-text leg and the dense leg **in parallel** and fuses +them with reciprocal rank fusion at `_RRF_K = 60`. `retrieval_mode` selects +`hybrid` (default), `vector_only` or `fts_only`; an unknown value falls back to +hybrid with a warning. Single-word FTS queries are rewritten to +`"* OR ~"` for prefix and fuzzy matching. Rows are de-duplicated on +`chunk_id` (then `_rowid`, then text) across the two legs, and each mode exposes +exactly one "higher is better" `score` field. + +**There are no fusion weights, and there is no knob to add them.** + +**Why.** BM25 scores are unbounded positives while cosine is bounded, so naive +weighted addition is meaningless; RRF sidesteps normalization entirely by using +ranks ([`component-map-2026.md` ยง4.3](research/component-map-2026.md)). Hybrid+RRF +measurably beats both legs (T2-RAGBench: Recall@5 0.695 vs BM25 0.644 vs dense +0.587), and โ€” the useful corrective โ€” **BM25 alone beat dense alone there by 5.7 +points**, so dropping the sparse leg is not the simplification it looks like. +Qdrant's own published guidance is the most honest vendor position available: +weighted RRF only when you have a tuned eval set with a train/val split, "neither +method dominates universally", and retune whenever retrievers, embeddings or +corpus change. ยง4.4's bottom line: "Do not spend time tuning fusion before you +have a reranker." This repo's history includes a *broken* weighted blend, which +RRF replaced ([`improvement_plan.md` ยง0](improvement_plan.md)). + +**What would change the decision.** A per-index place to store tuned weights plus +a per-corpus validation split โ€” that is [`improvement_plan.md`](improvement_plan.md) +ยง1.3, and it is correctly still open. The prerequisite is not the code; it is +having something to tune against that is not the same 72 queries used to evaluate. + +## 5. Evidence-sufficiency retry + +**What ships.** One conditional second retrieval +(`retrieval_pipeline.py::retrieve_candidates`), on in `default` +(`retrieval.retry: {enabled: True, min_top_score: 0.12, max_attempts: 1}`), off in +`fast`. When the first pass's evidence score falls below the threshold, the +enrichment model rewrites the query once (JSON-formatted so a small model's +"thinking" preamble cannot leak in), retrieval re-runs, and **the better of the +two result sets is kept** โ€” a retry that did not improve the evidence is +discarded, not merged. It is inert in `fts_only` mode and on legacy unnormalized +tables, by design: no signal, no retry. The retry surfaces as a +`retrieval_retry` SSE event. + +The signal is **not** the raw top similarity: + +``` +evidence = (cos_top โˆ’ cos_background) / (1 โˆ’ cos_background) +``` + +with `cos_background` the mean cosine from rank 6 down +(`_dense_evidence_score`). When a reranker returns a calibrated 0โ€“1 probability, +its top score is preferred (`_rerank_evidence_score`); arbitrary logits are +rejected rather than compared to a probability threshold. + +**Why.** The evidence says one conditional second iteration captures nearly all +of the deep-loop gain and that stopping criteria should be based on **cumulative +evidence sufficiency, not query count** โ€” across six agents on BrowseComp-Plus, +search volume correlates only weakly with answer quality, and redundant queries +characterise *underperforming* agents ([`component-map-2026.md` ยง8.6](research/component-map-2026.md)). +The broader 2026 rule is "escalate, don't pre-decide" +([`component-map-2026.md` ยง6.5, ยง7.4](research/component-map-2026.md)). + +Our own measurement changed the *signal*, which is the part worth recording. The +roadmap said to trigger on the top score. On the gold set, raw top cosine is +**anti-correlated with success**: all three `mixed` first-stage misses scored +*above* the median successful query, and any threshold catching all three fires +on 94% of the successes. Contrast works; absolute similarity mostly encodes how +close a query's phrasing sits to the corpus register +([`phase2-pipeline.md` ยง2.1](../eval/decisions/phase2-pipeline.md)). + +Effect, four runs: fires on **9.7โ€“11.1%** of `mixed`, **+0.008 to +0.017** +nDCG@10 and +0.014 recall@10, **zero per-query regressions**. It repaired one +genuine recall@10 = 0 miss. Mean latency over all queries 96 ms โ†’ 186 ms +([`phase2-pipeline.md` ยง2.3](../eval/decisions/phase2-pipeline.md)). + +**Honest limits, carried forward rather than buried.** The threshold was +calibrated on the same 72 queries it is evaluated on, against three first-stage +misses. `0.12` is a starting value, not a tuned constant; it catches **one** of +the three real misses, because catching all three costs a 26% false-fire rate. On +`mixed` the +0.008 is *inside* the one-query noise floor โ€” the reason to ship is +that it is positive on both corpora across four runs with nothing getting worse. + +**What would change the decision.** A larger gold set with more first-stage +misses to calibrate against. The firing rate drifted 9.7% โ†’ 11.1% as the corpus +grew, so it is a property of the corpus, not a constant โ€” re-check it after any +embedder change. + +## 5a. Per-query token accounting + +**What ships.** Every Ollama completion the agent makes โ€” streaming or not โ€” +reports `prompt_eval_count` and `eval_count` on its final object. Those are +aggregated per user query, bucketed by pipeline stage (`triage`, +`decomposition`, `synthesis`, `verification`), and returned as `token_usage` on +the `/chat` response body and in the SSE `complete` event. On by default: it +costs one dict update per LLM call and adds no request. + +The aggregation point is a `ContextVar` in `rag_system/utils/ollama_client.py` +rather than an argument threaded through every call site, because one +`OllamaClient` is shared by the agent, the retrieval pipeline, the verifier and +the decomposer. `await` and `asyncio.to_thread` propagate it; the agent's +parallel sub-query `ThreadPoolExecutor` copies it explicitly. + +Three honest gaps: the retry's reformulation call is billed to `synthesis`, +because it happens inside `RetrievalPipeline.run()` which the agent labels as +one stage; watsonx reports zeros, because the SDK path in use surfaces no +per-call counts; and an absent stage key means "no LLM call in that stage", +not "zero tokens". Evidence: `eval/decisions/phase4-escalation-tokens.md`. + +## 5b. Metadata filters and ask-a-folder + +**What ships.** `/chat` and `/chat/stream` accept an optional `filters` JSON +object (document id/name, chunk id, chunk-index ranges) compiled to LanceDB +where-clauses that prefilter **both** the vector and FTS legs. There is no +flag: with no `filters` argument the path is byte-identical to not having the +feature (md5-verified against a pre-change tree). Values containing quoting +characters are refused, never escaped; a malformed filter is a 400 from the one +validator in `rag_system/retrieval/filters.py`. Page and date filters are NOT +shipped โ€” they live inside the metadata JSON string column and need real +columns plus a re-index. + +`python -m rag_system.main ask ""` builds an ephemeral index +under a temp directory (fast profile, no enrichment), answers with the standard +pipeline, and removes everything afterwards โ€” including on SIGTERM. +Evidence: `eval/decisions/phase4-filters-askfolder.md`. + +## 6. Reranking posture + +**What ships.** Since arm G (2026-08-14) the `default` profile ships +`reranker.enabled = True` with threshold selection โ€” `top_k: 10`, +`min_score: 0.5`, `min_keep: 3` (`rag_system/main.py::PIPELINE_CONFIGS`), and +the UI "AI reranker" toggle defaults on to match +(`src/components/ui/session-chat.tsx`). `Qwen/Qwen3-Reranker-4B` is loaded +lazily on the first reranked query through the in-repo +`QwenRerankerScorer` (`rag_system/rerankers/reranker.py`), routed either by +explicit `reranker.model_type: "qwen3"` or by model name +(`retrieval_pipeline.py::_get_ai_reranker`). Any other model still goes through +the `rerankers` library. A reranker that fails to load logs a warning and is +skipped โ€” there is no fallback reranker. + +**Why this is the most counter-intuitive decision in the repo.** The evidence is +about as strong as evidence gets: reranking is "the single highest-ROI component +in the stack", +17.2 pp MRR@3 on T2-RAGBench, โˆ’1.7 EM when removed from a local +7B ablation, and the 2026 recommendation is a flat "**YES, unconditionally**" +([`component-map-2026.md` ยง5.1, ยง6.5](research/component-map-2026.md)). + +Our gold set says otherwise, for a specific and explicable reason. The joint +matrix ([`DECISIONS.md` ยง1](../eval/DECISIONS.md)), `mixed` corpus, same 20 +first-stage candidates reordered by each: + +| Stack | mixed nDCG@10 | docs nDCG@10 | added latency | +|---|---|---|---| +| **harrier-0.6b, first stage only** | 0.915 | 0.759 | โ€” | +| + `BAAI/bge-reranker-v2-m3` | 0.892 | 0.701 | **+1.6 s โ€” net negative** | +| + `Qwen/Qwen3-Reranker-4B` โ† shipped since arm G (2026-08-14) | 0.977 | 0.932 | +12.7 s | + +Two findings, both load-bearing: + +1. **The cheap cross-encoder now hurts.** bge-reranker-v2-m3's famous +0.232 on + `docs` was largely a *repair job on a weak first stage*. Improve the first + stage and the repair becomes damage. This is exactly why 1.1 and 1.2 could not + be decided independently and were re-measured jointly in one re-index window. +2. **The good reranker is a real win and was initially judged too slow to + default on.** +0.062 + nDCG@10 on `mixed`, +0.173 on `docs` โ€” the largest single quality win in this + repo's eval history โ€” for ~12.7 s per query and 7.5 GB of resident weights on + a single-user, single-threaded server. Arm G (2026-08-14) reversed the "off by + default" call: that call predated the synthesis context budget, when rank + order barely mattered because front-truncation fed synthesis the tail of the + list anyway. Now the budget keeps exactly the top-ranked documents, so + ordering and selection decide everything the model reads โ€” and `min_score` + selection sends a small, clean context on easy questions instead of a fixed + ten. + +`Qwen/Qwen3-Reranker-0.6B` was rejected outright: +0.021 nDCG@10 over bge (one to +two queries out of 72, inside the noise band) bought with 1.5โ€“2.8ร— the latency, +while *losing* recall@10 on both corpora ([`reranker.md` ยง7](../eval/decisions/reranker.md)). + +**One integration finding worth keeping.** `rerankers` 0.10.0 has no +Qwen3-Reranker backend. Loading one through the shipped cross-encoder path builds +a `Qwen3ForSequenceClassification` with a **randomly initialised score head** โ€” +had the batching not thrown, it would have returned untrained noise while +printing "AI reranker initialized successfully". `QwenRerankerScorer` implements +the model card's actual scheme (causal LM, left padding, chat template, +softmax over the `yes`/`no` logits at the final position), and the name-based +route in the loader exists specifically so no configuration can reach the random +head ([`reranker.md` ยง1](../eval/decisions/reranker.md)). + +**What would change the decision.** Three concrete triggers: + +* **Re-run the A/B if the embedder changes.** The reranker's headroom is largest + exactly where the first stage is weakest, so this decision is a function of ยง3 + and expires with it. +* **Tune the latency knobs.** Batch size (8) and the 2048-token truncation cap + are both untuned, and reranking only the top 10 candidates instead of 20 is + unmeasured. Either could move the 12.7 s materially โ€” that is unmeasured work, + not a promise. +* **A later `rerankers` release** may add a Qwen3 backend, at which point the + name-based route should be revisited. + +Note also what no reranker fixed: `docs_d09` and `docs_d17` degrade under every +model tested. They are a query-understanding problem, and this section is not +where they get solved. + +## 7. Query decomposition + +**What ships.** Since arm H (2026-08-15) the `default` profile ships +`compose_from_sub_answers: false` with `pooled_first_stage: true`: each +sub-query runs first-stage retrieval, the candidates are pooled and +de-duplicated, and there is ONE rerank pass and ONE synthesis over the union +context (`retrieval_pipeline.py::_pooled_first_stage`). Sub-queries are also +used at the *rerank* stage, where each candidate is scored against every +sub-query and the scores are aggregated by +`query_decomposition.rerank_aggregate` (`"mean"` default, `"max"` available) โ€” +`retrieval_pipeline.py::_rerank_stage`. The compose path โ€” a separate *answer* +per sub-question, composed into the final answer โ€” remains available behind +`compose_from_sub_answers: true` (`rag_system/agent/loop.py`). + +Consequence, stated plainly: the earlier posture โ€” **no shipped profile enables +sub-query scoring at rerank** โ€” held while the `default` profile kept the +reranker off and `compose_from_sub_answers: true`. Arm G turned the reranker on +(2026-08-14), so the aggregation path now runs by default, and arm H +(2026-08-15) made the pooled first stage the default decomposition path. + +**Why.** The evidence puts decomposition at "conditional โ€” multi-hop, applied at +*rerank*", noting that decomposition at initial retrieval dilutes the query +semantically ([`component-map-2026.md` ยง6.2, ยง6.5](research/component-map-2026.md)). + +We shipped the shape change and **measured the payload negative.** On `docs` with +`Qwen3-Reranker-4B`, only 6 of 24 queries decompose into more than one sub-query. +On exactly those 6, scoring against sub-queries at rerank is worse under both +aggregates: **0.8862 โ†’ 0.8406 (`max`, โˆ’0.046)** and **0.8740 (`mean`, โˆ’0.012)**. +The whole-corpus "gain" comes entirely from the 18 single-sub-query rows, where +the win is query *rewriting*, not decomposition +([`phase2-pipeline.md` ยง3](../eval/decisions/phase2-pipeline.md)). + +The shape change ships anyway because it is a strict reduction in work โ€” the +aggregate path used to issue N first-stage retrievals and now issues one โ€” and +because the structural check passed: the first-stage number is byte-identical +across all three arms, proving decomposition can no longer touch it. (Arm H's +pooled first stage later re-introduced per-sub-query retrieval; the reduction it +keeps is one rerank pass and one synthesis instead of N.) + +**What would change the decision.** n_effective = **6 queries on one corpus**. +That is too small to call the 2026 MultiConIR/SSRB finding wrong; it is big +enough to say it did not reproduce here, which is why nothing was switched on. A +multi-hop corpus with enough genuinely-decomposing queries to move a metric would +re-open it. `mean` stays the default aggregate on the "less bad" argument, not a +positive one. + +## 8. Routing / triage + +**What ships โ€” two layers, exactly one of which calls an LLM.** + +*Layer 1, the gateway* (`backend/server.py::should_use_rag`) is deterministic and +makes no network call: `force_rag` โ†’ RAG; no linked indexes โ†’ direct LLM; +whole-message smalltalk (โ‰ค 6 words, anchored allowlist, must contain a core +phrase) or assistant-meta ("who are you", "what model are you") โ†’ direct LLM; +**everything else โ†’ RAG.** Unit-tested at 155/155 in +`backend/test_gateway_routing.py`. + +*Layer 2, the agent* (`rag_system/agent/loop.py::_triage_query_async`) is the +system's single LLM routing layer: document overviews + the utility model decide +`rag_query` vs `direct_answer`, with "history exists โ†’ `rag_query`" as a shortcut +and an LLM fallback when no overviews are loaded. `_normalize_triage` collapses +anything that is not an explicit `direct_answer` to `rag_query`, so a small model +still emitting the retired `graph_query` label lands on the RAG path. + +**Why.** Pre-retrieval LLM routing is the weakest measured pattern of 2026, and +three independent sources agree: four ML approaches to pre-retrieval routing all +failed because "the need for augmentation cannot be determined from the query +alone"; rule-based retriever routing *lost* to fixed hybrid by 1.8 EM; and +TF-IDF+SVM matches or beats neural and LLM routers at ~zero cost +([`component-map-2026.md` ยง7.1, ยง7.3, ยง7.4](research/component-map-2026.md)). The +four-year pattern in ยง7.3 is "a small discriminative classifier is the right +tool; an LLM router is rarely justified." + +The bias toward over-sending to RAG is deliberate and cheap: agent triage runs on +every forwarded request and can still answer directly, so a false "use RAG" costs +one call on a model that would have been called anyway, while a false "answer +directly" costs an unanswerable question. The gateway is a smalltalk filter in +front of the decision-maker, not the decision-maker. + +**Measured** ([`phase2-gateway.md` ยง3](../eval/decisions/phase2-gateway.md)): mean +routing decision **750.613 ms โ†’ 0.002 ms** over the same 20 messages, with +**20/20 decision agreement** with the LLM router it replaced. The deleted +keyword fallback was worse than the old docs claimed โ€” it matched greetings by +*substring*, so `'hi'` matched *w**hi**ch*, *t**hi**s* and *mac**hi**ne*, routing +**7 of 8** real Atlas-7 questions to the direct LLM. That is why it was deleted +rather than patched, and why "messages containing test/check route RAG" is now a +regression test. + +**What would change the decision.** [`improvement_plan.md`](improvement_plan.md) +ยง2.1 (embed and cache document overviews for a cosine pre-check) and ยง2.2 +(session-level routing memo) both still stand for the *agent* layer, which is now +the only per-query LLM routing call. The evidence's own carve-out is that routing +still pays for *pipeline depth* selection, implemented as a post-retrieval +cascade โ€” which is what ยง5's retry is. + +## 9. Verification + +**What ships.** Verification is on in `default` (`verification: {enabled: True}`). +The shipped backend is an LLM prompt on the utility model +(`rag_system/agent/verifier.py::Verifier.verify_async`), returning a JSON verdict +and a confidence score; a low-confidence or ungrounded verdict appends a warning +to the answer (`agent/loop.py`). + +A **seam** exists for a local model: `VERIFIER_MODEL` / `verification.model` +swaps in `LocalNLIVerifier`, which sentence-splits the answer, scores each +sentence against the retrieved evidence as premise, and takes the **minimum** โ€” +one unsupported sentence makes the answer ungrounded, matching the binary +semantics `eval/judge.py` already uses. A model that cannot be loaded **raises**, +printing the availability table, rather than falling back: a verifier that +silently is not the verifier you configured is worse than an error. **The default +is unchanged.** + +**Why.** Verification helps as an *external* check, and the 2026 result is that a +4-bit 1B verifier (ThinknCheck, 78.1 BAcc) now beats the 7B 2024 SOTA, which +would make per-answer grounding cheap enough to always run +([`component-map-2026.md` ยง9.2](research/component-map-2026.md)). + +Both of the roadmap's named candidates failed availability checks against the +HuggingFace Hub API: **ThinknCheck has no public weights** (zero models returned; +the paper links no release), and Granite Guardian is either 8B/~16 GB or โ€” in its +38M form โ€” a hate/abuse classifier, the wrong task entirely. Two substitutes were +wired and exercised rather than left as a stub, and both were run against all 20 +hand-labelled cases in `eval/judge_validation.jsonl` +([`phase2-pipeline.md` ยง4](../eval/decisions/phase2-pipeline.md)): + +| Verifier | agreement | TPR | TNR | +|---|---|---|---| +| `lytang/MiniCheck-DeBERTa-v3-Large` (1.74 GB) | 19/20 | 10/10 | 9/10 | +| `MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli` (369 MB) | 18/20 | 8/10 | 10/10 | + +They fail in **opposite directions** โ€” MiniCheck missed a swapped-entity error, +the generic NLI model rejected two correct answers โ€” so neither is a drop-in +improvement without its own validation run. Nothing was made the default. + +**`[Confidence: N%]` is UX, not a calibrated measurement**, and +[`verifier.md`](verifier.md) says so in a callout. Changing the backend changes +where the number comes from; it does not calibrate it. The evidence supports the +caution independently: faithfulness metrics measure precision and ignore +coverage, so a system gated only on "is every claim supported" learns to abstain +([`component-map-2026.md` ยง9.4](research/component-map-2026.md)). + +**What would change the decision.** Run `eval/judge.py --validate`-style +TPR/TNR discipline on a candidate verifier over more than 20 cases, and beat the +LLM prompt on both directions rather than one. If ThinknCheck ever publishes +weights, it is the first thing to try. Also worth knowing: verifying against the +retrieved chunk alone systematically *under*-detects, because supporting evidence +is frequently outside the truncated passage +([`component-map-2026.md` ยง9.3](research/component-map-2026.md)). + +## 10. Sentence pruning + +**What ships.** Provence (`naver/provence-reranker-debertav3-v1`) via +`rag_system/rerankers/sentence_pruner.py`, applied after context expansion and +before synthesis (`retrieval_pipeline.py::run`), with fully-pruned chunks +dropped. It is **opt-in**: `provence.enabled` defaults to `False`, exposed as the +UI toggle "Prune irrelevant sentences" and as `provence_prune` on the API. The +model is loaded lazily, once, behind a lock, and a load failure skips pruning +rather than failing the query. + +**Why.** Provence is the one context-reduction variant whose economics survive +2026 scrutiny: it formulates pruning as sequence labelling **unified with +reranking**, so it is folded into a stage you already run, at "negligible to no +drop in performanceโ€ฆ at almost no cost" +([`component-map-2026.md` ยง10.1](research/component-map-2026.md)). ยง10.5's verdict +is exactly the posture here: "**prune, don't compress.**" + +It was kept off by default because localGPT did *not* run a cross-encoder by +default at the time (ยง6), so Provence's "zero marginal cost" argument โ€” the +entire reason it beats token-level compression โ€” did not hold in the shipped +configuration: an extra DeBERTa forward pass, not a free rider. Since arm G +(2026-08-14) the reranker **is** on by default, so that premise no longer holds; +folding pruning into the rerank stage is now a live candidate, but no `eval/` +number yet measures its effect on our gold set. + +**What would change the decision.** Measure it. Reranking returned to the +default profile in arm G (2026-08-14), so re-evaluate folding pruning in with +it. `OpenProvence` +(30Mโ€“310M, MIT, ModernBERT) is the lighter candidate the evidence points at. + +## 11. Caching and memory + +**What ships.** An in-process semantic cache in the agent +(`rag_system/agent/loop.py`): the raw query is embedded, compared by cosine +against cached entries, and a hit at or above `semantic_cache_threshold` +(**0.98**) returns the stored result. `cache_scope` defaults to **`"session"`**, +so an entry from another session is skipped before the similarity is even +computed. Conversation memory is a session transcript โ€” an in-process +`chat_histories` map plus SQLite rows in `backend/chat_data.db` โ€” formatted into +the query as history. There is **no vendor memory layer**, no entity graph, no +consolidation pass. + +**Why.** + +* **The null baseline wins.** The 2026 guidance is explicit: "plain RAG over a + session transcript store as the baseline you must beat" โ€” Cloud-RAG beat Mem0 + on LongMemEval-S, a filesystem agent beat Mem0 on LoCoMo, and the benchmark the + whole category is scored on (LOCOMO) is broadly discredited on structural + grounds ([`component-map-2026.md` ยง11.1, ยง11.4](research/component-map-2026.md)). + Our session store *is* that null baseline. The same section notes a bare + embedding-model swap moves accuracy 6.2 pp โ€” **larger than most claimed + memory-system deltas** โ€” which is where the effort went instead (ยง3). +* **Session scoping is a privacy fix, not a performance one.** A global cache + returns one session's answer to another session's user. + [`improvement_plan.md` ยง0](improvement_plan.md) records it as closing a + cross-session answer leak. +* **The 0.98 threshold is deliberately extreme.** Semantic caches are attackable + by embedding-similarity collision (86% hijacking hit rate), and the consistent + 2026 theme is that hit rate is the wrong headline metric โ€” calibration, + freshness and collision-resistance decide whether a cache is *safe* + ([`component-map-2026.md` ยง11.3](research/component-map-2026.md)). At 0.98 + nothing but a near-identical query hits, which was also verified as a side + effect of the prefix work: with the query prefix on, mean pairwise cosine over + the 72 gold queries *falls* (0.361 โ†’ 0.216) and max reaches 0.899, so nothing + crosses the bar either way ([`embedder.md` ยง8](../eval/decisions/embedder.md)). + +**What would change the decision.** Instrument the cache for calibration and +staleness rather than hit rate before loosening the threshold. Graphiti + +FalkorDB Lite is the evidence's pick *if* temporal/entity memory with provenance +is ever actually needed โ€” but the null baseline has to lose on our own eval first, +and there is no memory eval in `eval/` today. + +## 12. Streaming and persistence + +**What ships.** `POST :8001/chat/stream` emits SSE +(`rag_system/api_server.py::handle_chat_stream`) and every pipeline phase is a +typed event: `retrieval_started`, `retrieval_done`, `retrieval_retry`, +`rerank_started`, `rerank_done`, `context_expand_*`, `prune_*`, `token`, +`sub_query_token`, `complete`. Synthesis streams token-by-token from +`_synthesize_final_answer`. A client disconnect (`BrokenPipeError`) is logged, +not raised. + +**The stream writes nothing to SQLite.** When the stream finishes, the UI posts +the completed turn to `POST :8000/sessions/{id}/messages/save` +(`src/lib/api.ts::saveStreamedTurn`, `backend/server.py`), which persists the +user message, the assistant message, the source documents and the pipeline +cascade in `metadata.steps`. Direct stream consumers must do the same to get +history โ€” documented as known limitation #1 in +[`system_overview.md`](system_overview.md#11-known-limitations). + +**Why.** This is an architecture choice, not an evidence-backed one, and it is +recorded here so it is not mistaken for the latter. The rationale is +single-writer discipline: the RAG API owns retrieval and the vector store, the +gateway owns SQLite, and no request writes to a store it does not own. The cost +is the extra round-trip and the fact that an interrupted stream persists nothing. + +The one thing the evidence *does* motivate is that the retry and the rerank are +surfaced as SSE steps rather than hidden โ€” the 2026 escalation-architecture +literature treats visible, staged escalation as the point of the design +([`component-map-2026.md` ยง8.6](research/component-map-2026.md), PEA-CAE). + +**Verified by** `eval/smoke_e2e.py`: **25/25 assertions**, including planted-fact +answers with non-empty `source_documents`, the `[Confidence: N%]` tag, and the +streamed-turn save/round-trip ([`phase2-gateway.md` ยง3.4](../eval/decisions/phase2-gateway.md)). +Note the honest gap recorded there: smoke sends `force_rag: True`, so it +exercises the gateway's `force_rag` branch, not the discriminative one. + +**What would change the decision.** Making the stream itself durable would need +the RAG API to write session state, which crosses the ownership boundary. The +lower-cost fix is a server-side save on stream completion in the gateway's +proxy path. + +--- + +## 13. Deliberately not implemented + +Each of these is **absent on purpose**. If you are about to add one, read the row +first โ€” and bring a gold-set number, not a paper. + +| Not implemented | Why (evidence) | Revisit when | +|---|---|---| +| **HyDE** โ€” in any form. `grep -i hyde` over `rag_system/`, `backend/`, `src/` and `eval/` returns **zero hits**; there is no flag, no dead code, no "coming soon". | HyDE underperforms plain dense retrieval on entity/numeric corpora (T2-RAGBench: Recall@5 0.544 vs 0.587, nDCG@10 0.433 vs 0.466) because pseudo-documents fabricate figures; and in a production system only **27.8%** of real queries needed LLM augmentation while synthetic evals implied >90% โ€” "the Coverage Illusion". Verdict: "alive but no longer a default. **Never always-on.**" ([`component-map-2026.md` ยง6.1](research/component-map-2026.md)) | Only as a *post-retrieval* escalation (retrieval returned nothing โ†’ then HyDE), or moved to index time (HyPE). The escalation slot in this codebase is already occupied by ยง5's retry; a HyDE variant would have to beat it on the gold set. | +| **Multi-query expansion** | The weakest of the three query transformations: on T2-RAGBench multi-query scored Recall@5 0.640 โ€” **worse than plain BM25** (0.644). Prompt-only LLM rewriting measured **โˆ’9.0% nDCG@10 (p<0.001)** on FiQA, and an attempt to *gate* the rewriter reached only AUC 0.593 ([`component-map-2026.md` ยง6.3](research/component-map-2026.md)). | A trained lexical expander with BM25-level cost (the STORM line) that wins on our gold set. Note ยง5's retry is a *single conditional* rewrite kept only when it scores better โ€” that is the sanctioned shape. | +| **Weighted / tunable fusion knobs** | BM25 and cosine are not on a common scale; "a fixed alpha over raw scores tends to be dominated by whichever retriever has larger raw magnitudes", and no 2026 evidence shows learned fusion beating RRF in the general case ([`component-map-2026.md` ยง4.3, ยง4.4](research/component-map-2026.md)). RRF is the scale-free safe default. | A per-index validation split exists to tune against and a place to store per-corpus weights โ€” [`improvement_plan.md`](improvement_plan.md) ยง1.3. Tuning fusion before a reranker is explicitly the wrong order. | +| **GraphRAG** โ€” removed 2026-08-09 | Loses on single-hop retrieval (64.78 vs 63.01 F1 on NQ); multi-hop gains span +3 to +27 depending entirely on how well the vector baseline is tuned; costs **41โ€“57ร— at indexing** and up to **~377ร— in query tokens** ([`academic-evidence-2026.md` ยง6](research/academic-evidence-2026.md)). It was also unreachable โ€” no shipped profile ever set `graph_strategy`. | A corpus where entity-linking is the task and a tuned vector baseline demonstrably fails on our gold set. Deleted: `graph_extractor.py`, `GraphRetriever`, `GraphQueryTranslator`, the `graph_query` triage outcome, and `networkx`/`fuzzywuzzy`/`python-Levenshtein` ([`phase2-pipeline.md` ยง1](../eval/decisions/phase2-pipeline.md)). | +| **Vendor memory systems** (Mem0, Zep/Graphiti, Letta, LangMem, GPTCache) | "Plain RAG over a session transcript store [is] the baseline you must beat" โ€” Cloud-RAG beat Mem0 on LongMemEval-S, a filesystem agent beat it on LoCoMo, and LOCOMO itself is discredited (conversations of 16kโ€“26k tokens do not stress memory). GPTCache is dormant since Aug 2024 ([`component-map-2026.md` ยง11.1, ยง11.2, ยง11.4](research/component-map-2026.md)). | The null baseline (ยง11) loses on a memory eval we actually own. Fix the embedding model before comparing memory systems โ€” a bare swap moves accuracy 6.2 pp, more than most claimed memory deltas. | +| **Deep subagent / parallel fan-out loops** | The pro case (Anthropic's orchestrator-worker, +90.2%) runs at **~15ร— chat tokens** on frontier models. The 2026 counter-evidence: on repository-level code QA, plain semantic search scored **65.2%** vs deep agentic search **46.2% at >2ร— cost**, with **41.8% of failures at the plannerโ†’subagent hand-off** โ€” "usually silent, ending in a fluent and confident answer that was wrong" ([`component-map-2026.md` ยง8.6](research/component-map-2026.md)). | Never, for a read-only indexable corpus on a single-user local box. The bounded version of this idea is already shipped: one conditional retry (ยง5), capped at `max_attempts: 1`. | +| **RL-trained searchers** (Search-R1 lineage) | On BrowseComp-Plus โ€” a fixed, human-verified corpus that disentangles retriever from agent โ€” **Search-R1 + BM25 scores 3.86%** while GPT-5 + BM25 scores 55.9% and GPT-5 + Qwen3-Embedding-8B scores 70.1%, with *fewer* search calls (ACL 2026 Main, [`component-map-2026.md` ยง8.3](research/component-map-2026.md)). No out-of-distribution transfer; also mis-calibrates confidence, which would corrupt ยง5's evidence signal. | A local-training story with demonstrated OOD transfer. ยง8.2 documents the 2026 lineage if that changes; nothing in it is currently a better use of a laptop GPU than a better embedder. | +| **Token-level context compression** (LLMLingua-style) | Across thousands of runs on 30,000 queries and three GPU classes: **at most ~18% end-to-end speedup**, and only when prompt length, ratio and hardware align โ€” outside that window "compression overhead dominates and cancels the decoding gains entirely" (ECIR 2026). Fixed ratios reversed **31%** of pairwise model rankings on LongMemEval-S and obscured **80%** of the gain from a reader upgrade; hard compressors leave the answer path incomplete in 34โ€“60% of multi-hop bridge examples. Upstream quiet since Dec 2024 ([`component-map-2026.md` ยง10.2โ€“10.5](research/component-map-2026.md)). | Not in this shape. The surviving alternative โ€” extractive sentence pruning โ€” is already shipped as ยง10. If compression is ever profiled here, do it per model/hardware pair and re-run the ablation on every reader upgrade. | + +### 13a. Implemented but measured out of the defaults (Phase 4, 2026-08-09) + +These three shipped as flag-gated code, were benchmarked on/off, and stay OFF on +the numbers. The flags exist so the A/Bs can be re-run; the code is not dead, it +is *disabled by measurement*. + +| Flag (both profiles) | Verdict | The number that decided it | +|---|---|---| +| `retrieval.document_escalation` (4.1) | **REJECTED as default** (was HOLD) | The 2026-08-09 lift (0/7โ†’2/7 judged on fired queries) was the truncation bug, not document reassembly: the escalated block was appended at the tail and *survived* front-truncation while top-ranked chunks were discarded. The HOLD's condition โ€” re-run after the context-window fix โ€” was executed 2026-08-12 (`eval/decisions/phase4-escalation-rerun.md`): with prompts that fit (249 calls, zero truncation), the escalation-OFF baseline on the identical fire subset went 0/9 โ†’ 7/9 and escalation added nothing on top (5/7 โ†’ 2/7 mechanically, 5/7 โ†’ 5/7 by hand adjudication); both fires under the true product default were regressions on both dates. Flag and code kept โ€” unfalsified for a corpus where document ordering matters within a fitting prompt; this one never presents that case. | +| `retrieval.crossref_hop` (4.2) | **REJECTED as a default** | Fires 0/11 at the shipped k=20; where forced (k=5), 0/11 hopped chunks hit an expected source โ€” target selection is query-blind and lands on hub documents (21/24 hops to 2 documents) โ€” and raising `k` beats the hop at equal context budget in 3 of 4 cells. Index-time `extract_crossrefs` stays ON (free, regex-only, bit-identical text/vectors). `eval/decisions/phase4-retrieval-benchmarks.md`, `phase4-answer-quality.md`. | +| `retrieval.overview_prefilter` (4.3) | **boost: HOLD ยท restrict: REJECTED** | boost: +0.106 nDCG@10 on the heterogeneous acquisition slice, โˆ’0.021 on `mixed` โ€” a per-index opt-in candidate, not a default. restrict: removed the answer document entirely (recall@20 1โ†’0) for 4 queries per corpus. `eval/decisions/phase4-retrieval-benchmarks.md`. | + +## 14. How to re-litigate a decision + +None of the above is permanent. The process for changing one: + +1. **Reproduce the recorded number first.** The `docs` and `mixed` corpora index + live `Documentation/*.md` content, so **editing documentation moves the + metric.** Snapshot the pre-change tree and record the chunk count; only + compare runs made against the same tree. + + ```bash + .venv/bin/python eval/run_eval.py --corpus all \ + --json-out eval/results/before.json + ``` + +2. **Run your change against the same corpus snapshot**, and run it twice if any + arm makes an LLM call (the retry does), so the spread between runs is visible + rather than mistaken for signal. + + ```bash + .venv/bin/python eval/run_eval.py --corpus all \ + --json-out eval/results/after.json + .venv/bin/python eval/smoke_e2e.py # 25/25 assertions, exit 0 + ``` + +3. **Beat the recorded number by more than the noise floor.** On `mixed`, one + query โ‰ˆ **0.014 nDCG@10**. Deltas under that are not findings. A delta that is + positive on both corpora across multiple runs with zero per-query regressions + is (that is the standard ยง5 was held to). Public-leaderboard deltas under ~2 + points are known not to transfer + ([`component-map-2026.md` ยง3.6](research/component-map-2026.md)). + +4. **Write it down.** Add the outcome and the numbers to + [`eval/DECISIONS.md`](../eval/DECISIONS.md), with a page under + [`eval/decisions/`](../eval/decisions/) if the investigation is substantial. + Record the *rejections* too โ€” the negative results in + [`reranker.md`](../eval/decisions/reranker.md) and + [`phase2-pipeline.md`](../eval/decisions/phase2-pipeline.md) are the most + valuable pages in that directory. + +5. **Land code, docs and the eval delta in the same change**, and update the + relevant section here. Nothing in this file may describe unshipped behaviour โ€” + that is what [`research_roadmap.md`](research_roadmap.md) is for. + +Two standing rules from the harness itself: never quietly repair a gold row to +make your own change look better (`docs_d10` was left failing and reported until +the gate re-anchored it, on the record), and never quote a leaderboard position in +this repo's docs โ€” cite our own eval or nothing. diff --git a/Documentation/docker_usage.md b/Documentation/docker_usage.md index 101307fb..4e017207 100644 --- a/Documentation/docker_usage.md +++ b/Documentation/docker_usage.md @@ -1,101 +1,126 @@ -# ๐Ÿณ Docker Usage Guide - RAG System +# ๐Ÿณ Docker Usage Guide - LocalGPT -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ -This guide provides practical Docker commands and procedures for running the RAG system in containerized environments with local Ollama. +Practical Docker commands and procedures for running LocalGPT in containers. --- ## ๐Ÿ“‹ Prerequisites ### Required Setup -- Docker Desktop installed and running -- Ollama installed locally (even for Docker deployment) +- Docker Desktop (or Docker Engine 24+ with the Compose plugin) running +- Ollama, either on the host (default) or as a container - 8GB+ RAM available ### Architecture Overview ``` -โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ” -โ”‚ Docker Containers โ”‚ -โ”œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”ค -โ”‚ Frontend (Port 3000) โ”‚ -โ”‚ Backend (Port 8000) โ”‚ -โ”‚ RAG API (Port 8001) โ”‚ -โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ - โ”‚ - โ–ผ -โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ” -โ”‚ Local System โ”‚ -โ”œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”ค -โ”‚ Ollama Server (Port 11434) โ”‚ -โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ +โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ” +โ”‚ Docker Containers โ”‚ +โ”œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”ค +โ”‚ frontend (3000) โ†’ backend (8000) โ†’ rag-api โ”‚ +โ”‚ (8001) โ”‚ +โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ + โ”‚ โ”‚ + โ”‚ browser streams โ”‚ + โ”‚ directly to :8001 โ–ผ + โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ–บ โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ” + โ”‚ Ollama (11434) โ”‚ + โ”‚ host.docker.internalโ”‚ + โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ ``` +`backend` and `rag-api` share `./backend` (SQLite), `./lancedb`, `./index_store` +and `./shared_uploads` as bind mounts. + --- ## 1. Quick Start Commands -### Step 1: Clone and Setup +### Step 1: Clone ```bash -# Clone repository -git clone -cd rag_system_old +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Verify Docker is running docker version ``` -### Step 2: Install and Configure Ollama (Required) +### Step 2: Ollama -**โš ๏ธ Important**: Even with Docker, Ollama must be installed locally for optimal performance. +The compose files point the containers at `host.docker.internal:11434` by default, +and declare `extra_hosts: ["host.docker.internal:host-gateway"]` so that also +resolves on Linux. ```bash -# Install Ollama +# Install Ollama on the host curl -fsSL https://ollama.ai/install.sh | sh # Start Ollama (in one terminal) ollama serve -# Install required models (in another terminal) -ollama pull qwen3:0.6b # Fast model (650MB) -ollama pull qwen3:8b # High-quality model (4.7GB) +# Install the models (in another terminal) +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, enrichment, verification -# Verify models are installed ollama list - -# Test Ollama connection curl http://localhost:11434/api/tags ``` -### Step 3: Start Docker Containers +**Or run Ollama in a container.** `docker-compose.yml` defines an `ollama` service +behind the `with-ollama` profile: ```bash -# Start all containers -./start-docker.sh - -# Stop all containers -./start-docker.sh stop +./start-docker.sh container +# equivalently: +# OLLAMA_HOST=http://ollama:11434 \ +# docker compose --env-file docker.env --profile with-ollama up --build -d + +# Models must be pulled inside the container the first time +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` -# View logs -./start-docker.sh logs +The embedding model (`microsoft/harrier-oss-v1-0.6b`) is a HuggingFace download +inside `rag-api`, not an Ollama model โ€” nothing to pull for it. The reranker +(`Qwen/Qwen3-Reranker-4B`, ~7.5 GB) loads lazily: reranking is on by default, so +it is downloaded on the first reranked query, not at startup. -# Check status -./start-docker.sh status +### Step 3: Start Containers -# Restart containers +```bash +./start-docker.sh # local Ollama +./start-docker.sh container # containerized Ollama ./start-docker.sh stop -./start-docker.sh +./start-docker.sh logs +./start-docker.sh status +./start-docker.sh help ``` +`./start-docker.sh` with no argument probes port 11434. If nothing is listening it +offers the containerized fallback; pass `-y`/`--yes` (or set `NONINTERACTIVE=1`) to +take it without a prompt, which is what CI should do. With no TTY and no `-y` it +exits 1 with instructions instead of hanging. + ### 1.2 Service Access -Once running, access the system at: - **Frontend**: http://localhost:3000 -- **Backend API**: http://localhost:8000 +- **Backend API**: http://localhost:8000 - **RAG API**: http://localhost:8001 - **Ollama**: http://localhost:11434 +### 1.3 Startup Order + +``` +rag-api (healthy) โ†’ backend (healthy) โ†’ frontend +``` + +`rag-api` loads the embedding model before `/health` answers (the reranker loads +lazily, on the first reranked query), so its +check has a 120s start period and `backend` intentionally waits in `created` until +then. `docker compose logs -f rag-api` shows the progress. + --- ## 2. Container Management @@ -103,21 +128,13 @@ Once running, access the system at: ### 2.1 Using the Convenience Script ```bash -# Start all containers ./start-docker.sh - -# Stop all containers ./start-docker.sh stop - -# View logs ./start-docker.sh logs - -# Check status ./start-docker.sh status -# Restart containers -./start-docker.sh stop -./start-docker.sh +# Restart +./start-docker.sh stop && ./start-docker.sh ``` ### 2.2 Manual Docker Compose Commands @@ -137,24 +154,22 @@ docker compose down # Force rebuild docker compose build --no-cache -docker compose up --build -d +docker compose --env-file docker.env up --build -d ``` +Always pass `--env-file docker.env` (as `start-docker.sh` does). Without it the +compose defaults still apply โ€” they mirror `docker.env` โ€” but any value you edit in +`docker.env` is ignored. + ### 2.3 Individual Service Management ```bash -# Start specific service docker compose up -d frontend docker compose up -d backend docker compose up -d rag-api -# Restart specific service docker compose restart rag-api - -# Stop specific service docker compose stop backend - -# View specific service logs docker compose logs -f rag-api ``` @@ -164,46 +179,48 @@ docker compose logs -f rag-api ### 3.1 Code Changes -```bash -# After frontend changes -docker compose restart frontend - -# After backend changes -docker compose restart backend +No source is bind-mounted โ€” code is `COPY`-ed into the images at build time, so a +restart alone will not pick up an edit. Rebuild the affected service: -# After RAG system changes -docker compose restart rag-api +```bash +docker compose up -d --build frontend +docker compose up -d --build backend +docker compose up -d --build rag-api -# Rebuild after dependency changes +# After a dependency change docker compose build --no-cache rag-api docker compose up -d rag-api ``` +Editing `NEXT_PUBLIC_API_URL` or `NEXT_PUBLIC_RAG_API_URL` also needs a frontend +**rebuild** โ€” Next.js inlines those at `next build` time and they are passed to +`Dockerfile.frontend` as build args. + ### 3.2 Debugging Containers ```bash -# Access container shell -docker compose exec frontend sh -docker compose exec backend bash -docker compose exec rag-api bash +# Shells +docker compose exec frontend sh # node:20-alpine +docker compose exec backend bash # python:3.11-slim +docker compose exec rag-api bash # python:3.11-slim -# Run commands in container -docker compose exec rag-api python -c "from rag_system.main import get_agent; print('โœ… RAG System OK')" -docker compose exec backend curl http://localhost:8000/health +# Run commands in a container +docker compose exec rag-api python -c "from rag_system.factory import get_agent; get_agent('default'); print('โœ… RAG System OK')" +docker compose exec backend curl -s http://localhost:8000/health -# Check environment variables -docker compose exec rag-api env | grep OLLAMA +# Environment +docker compose exec rag-api env | grep -E "OLLAMA|MODEL|DB_PATH|LANCEDB" ``` -### 3.3 Development vs Production +### 3.3 Compose File Variants -```bash -# Development mode (if docker-compose.dev.yml exists) -docker compose -f docker-compose.yml -f docker-compose.dev.yml up -d +| File | Contents | +|------|----------| +| `docker-compose.yml` | `rag-api`, `backend`, `frontend`, plus an optional `ollama` service behind the `with-ollama` profile | -# Production mode (default) -docker compose --env-file docker.env up -d -``` +This is the only compose file (the old `docker-compose.local-ollama.yml` variant +was removed; `--profile with-ollama` covers that flow). There is no +`docker-compose.dev.yml`. --- @@ -212,131 +229,123 @@ docker compose --env-file docker.env up -d ### 4.1 Log Management ```bash -# View all logs docker compose logs - -# View specific service logs docker compose logs frontend docker compose logs backend docker compose logs rag-api -# Follow logs in real-time docker compose logs -f - -# View last N lines docker compose logs --tail=100 - -# View logs with timestamps docker compose logs -t - -# Save logs to file docker compose logs > system.log 2>&1 - -# View logs since specific time docker compose logs --since=2h -docker compose logs --since=2025-01-01T00:00:00 ``` ### 4.2 System Monitoring ```bash -# Monitor resource usage docker stats - -# Monitor specific containers docker stats rag-frontend rag-backend rag-api -# Check container health docker compose ps +docker inspect rag-api --format='{{.State.Health.Status}}' -# System information docker system info docker system df ``` +Container names are fixed by `container_name`: `rag-frontend`, `rag-backend`, +`rag-api`, and `rag-ollama` for the optional Ollama service. + --- ## 5. Ollama Integration -### 5.1 Ollama Setup +### 5.1 Host Ollama ```bash -# Install Ollama (one-time setup) curl -fsSL https://ollama.ai/install.sh | sh - -# Start Ollama server ollama serve - -# Check Ollama status curl http://localhost:11434/api/tags -# Install models -ollama pull qwen3:0.6b # Fast model -ollama pull qwen3:8b # High-quality model - -# List installed models +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ollama list ``` -### 5.2 Ollama Management +### 5.2 From Inside a Container ```bash -# Check model status from container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +docker compose exec rag-api curl -s http://host.docker.internal:11434/api/tags -# Test Ollama connection curl -X POST http://localhost:11434/api/generate \ -H "Content-Type: application/json" \ - -d '{"model": "qwen3:0.6b", "prompt": "Hello", "stream": false}' - -# Monitor Ollama logs (if running with logs) -# Ollama logs appear in the terminal where you ran 'ollama serve' + -d '{"model": "qwen3.5:4b", "prompt": "Hello", "stream": false}' ``` +Ollama logs appear in the terminal running `ollama serve`; for the containerized +variant use `docker compose --profile with-ollama logs -f ollama`. + ### 5.3 Model Management ```bash -# Update models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b +ollama pull qwen3.6:27b # optional high-end generation model -# Remove unused models ollama rm old-model-name +ollama show qwen3.5:9b +``` + +Point the containers at a different model without editing code: -# Check model information -ollama show qwen3:0.6b +```bash +GENERATION_MODEL=qwen3.6:27b docker compose --env-file docker.env up -d rag-api backend ``` --- ## 6. Data Management -### 6.1 Volume Management +### 6.1 Volumes and Mounts + +Every application path is a **bind mount to a host directory**, so ordinary file +tools work. The only named volume is `ollama_data`, used by the optional Ollama +container. + +| Host path | Container path | Contents | +|-----------|----------------|----------| +| `./lancedb` | `/app/lancedb` | Vectors and the native full-text index | +| `./index_store` | `/app/index_store` | Document overviews | +| `./shared_uploads` | `/app/shared_uploads` | Uploaded source documents | +| `./backend` | `/app/backend` | `chat_data.db` (shared by backend and rag-api) | ```bash -# List volumes docker volume ls - -# View volume usage docker system df -v -# Backup volumes -docker run --rm -v rag_system_old_lancedb:/data -v $(pwd)/backup:/backup alpine tar czf /backup/lancedb_backup.tar.gz -C /data . +# Back up the host directories directly +tar czf backup/lancedb_backup.tar.gz lancedb/ +tar czf backup/index_store_backup.tar.gz index_store/ + +# Only the containerized Ollama uses a named volume. +# The compose project name is the directory name, so it is localgpt_ollama_data. +docker run --rm -v localgpt_ollama_data:/data -v $(pwd)/backup:/backup \ + alpine tar czf /backup/ollama_models.tar.gz -C /data . -# Clean unused volumes docker volume prune ``` ### 6.2 Database Management ```bash -# Access SQLite database -docker compose exec backend sqlite3 /app/backend/chat_data.db +# sqlite3 is installed in both Python images +docker compose exec backend sqlite3 /app/backend/chat_data.db ".tables" -# Backup database +# Back up the database cp backend/chat_data.db backup/chat_data_$(date +%Y%m%d).db -# Check LanceDB tables from container +# Check LanceDB tables from the container docker compose exec rag-api python -c " import lancedb db = lancedb.connect('/app/lancedb') @@ -344,20 +353,23 @@ print('Tables:', db.table_names()) " ``` +`backend` and `rag-api` both set `DB_PATH=/app/backend/chat_data.db` and mount the +same host directory, so they read and write one file. + ### 6.3 File Management ```bash -# Access shared files docker compose exec rag-api ls -la /app/shared_uploads -# Copy files to/from containers docker cp local_file.pdf rag-api:/app/shared_uploads/ docker cp rag-api:/app/shared_uploads/file.pdf ./local_file.pdf -# Check disk usage docker compose exec rag-api df -h ``` +Because `shared_uploads/` is a bind mount, copying a file into `./shared_uploads` +on the host is equivalent and simpler. + --- ## 7. Troubleshooting @@ -366,43 +378,42 @@ docker compose exec rag-api df -h #### Container Won't Start ```bash -# Check Docker daemon docker version - -# Check for port conflicts lsof -i :3000 -i :8000 -i :8001 - -# Check container logs docker compose logs [service-name] - -# Restart Docker Desktop -# macOS/Windows: Restart Docker Desktop -# Linux: sudo systemctl restart docker ``` +#### `backend` stays in `created` +That is `depends_on: rag-api: condition: service_healthy` doing its job. Watch +`docker compose logs -f rag-api` โ€” on a cold start it is downloading and loading +the embedding model (the reranker loads lazily later, on the first reranked +query). + #### Ollama Connection Issues ```bash -# Check Ollama is running curl http://localhost:11434/api/tags -# Restart Ollama pkill ollama ollama serve -# Check from container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +docker compose exec rag-api curl -s http://host.docker.internal:11434/api/tags +``` + +#### Chats return "Could not connect to the RAG API server" +The backend builds its URLs from `RAG_API_URL`, which must be +`http://rag-api:8001` inside compose (`localhost` there means the backend +container itself). +```bash +docker compose exec backend env | grep RAG_API_URL +docker compose exec backend curl -s http://rag-api:8001/health ``` #### Performance Issues ```bash -# Check resource usage docker stats - -# Increase Docker memory (Docker Desktop Settings) -# Recommended: 8GB+ for Docker - -# Check container health docker compose ps + +# Docker Desktop โ†’ Settings โ†’ Resources โ†’ Memory โ†’ 8GB+ ``` ### 7.2 Reset and Clean @@ -414,34 +425,37 @@ docker compose ps # Clean containers and images docker system prune -a -# Clean volumes (โš ๏ธ deletes data) -docker volume prune - -# Complete reset (โš ๏ธ deletes everything) -docker compose down -v -docker system prune -a --volumes +# Complete reset (โš ๏ธ deletes indexes, uploads and chat history) +docker compose down +rm -rf lancedb/* index_store/* shared_uploads/* backend/chat_data.db +docker system prune -a ``` +`docker compose down -v` only removes the named `ollama_data` volume โ€” application +data lives in host directories and must be deleted explicitly. + ### 7.3 Health Checks ```bash -# Comprehensive health check curl -f http://localhost:3000 && echo "โœ… Frontend OK" curl -f http://localhost:8000/health && echo "โœ… Backend OK" -curl -f http://localhost:8001/models && echo "โœ… RAG API OK" +curl -f http://localhost:8001/health && echo "โœ… RAG API OK" curl -f http://localhost:11434/api/tags && echo "โœ… Ollama OK" -# Check all container status docker compose ps -# Test model loading +# Test model loading inside the container docker compose exec rag-api python -c " -from rag_system.main import get_agent +from rag_system.factory import get_agent agent = get_agent('default') print('โœ… RAG System initialized successfully') " ``` +These are the same endpoints the container health checks use: `curl -f /health` +for `backend` and `rag-api`, and busybox `wget -qO- http://localhost:3000` for +`frontend` (the alpine image has no curl). + --- ## 8. Advanced Usage @@ -449,36 +463,37 @@ print('โœ… RAG System initialized successfully') ### 8.1 Production Deployment ```bash -# Use production environment -export NODE_ENV=production - -# Start with resource limits -docker compose --env-file docker.env up -d +# docker.env already sets NODE_ENV=production +docker compose --env-file docker.env up --build -d -# Enable automatic restarts -docker update --restart unless-stopped $(docker ps -q) +# All services already declare restart: unless-stopped +docker compose ps ``` +There is no authentication and CORS is wide open on both APIs. Put a reverse proxy +in front and publish only what you need. Note the browser streams directly from +port 8001, so that port must be reachable by clients (or proxied) unless you turn +off "Stream phases" in the chat UI. + ### 8.2 Scaling -```bash -# Scale specific services -docker compose up -d --scale backend=2 --scale rag-api=2 +`docker compose up -d --scale backend=2 --scale rag-api=2` **does not work with the +shipped compose file**: each service sets a fixed `container_name` +(`rag-backend`, `rag-api`) and publishes a fixed host port, both of which conflict +on the second replica. Remove `container_name` and the `ports:` mappings (or switch +to a random host port) first. -# Use Docker Swarm for clustering -docker swarm init -docker stack deploy -c docker-compose.yml rag-system -``` +Even then, the RAG API is single-threaded and holds per-process state โ€” the +resident agent, its in-memory chat history and the semantic cache โ€” so replicas +need sticky sessions at minimum. ### 8.3 Security ```bash -# Scan images for vulnerabilities docker scout cves rag-frontend docker scout cves rag-backend docker scout cves rag-api -# Update base images docker compose build --no-cache --pull ``` @@ -488,56 +503,75 @@ docker compose build --no-cache --pull ### 9.1 Environment Variables -The system uses `docker.env` for configuration: +`docker.env` is passed with `--env-file` and supplies both runtime environment and +compose-level substitution: ```bash -# Ollama configuration +# Ollama on the host; extra_hosts makes this resolve on Linux too OLLAMA_HOST=http://host.docker.internal:11434 +# Containerized alternative: OLLAMA_HOST=http://ollama:11434 -# Service configuration NODE_ENV=production RAG_API_URL=http://rag-api:8001 + +# Browser-facing; inlined into the frontend bundle at build time NEXT_PUBLIC_API_URL=http://localhost:8000 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 + +# Shared SQLite + vector store +DB_PATH=/app/backend/chat_data.db +LANCEDB_PATH=/app/lancedb + +# Models +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B ``` +Values already exported in your shell win over `--env-file` โ€” that is how +`start-docker.sh container` overrides `OLLAMA_HOST`. + +Changing `EMBEDDING_MODEL` invalidates existing indexes: every table records the +embedding model that wrote it (and its vector width), and writing or querying it +with a different model fails with an explicit error. Rebuild your indexes after +switching. + ### 9.2 Custom Configuration ```bash -# Create custom environment file cp docker.env docker.custom.env - -# Edit custom configuration nano docker.custom.env - -# Use custom configuration docker compose --env-file docker.custom.env up -d + +# Remember to rebuild if you changed a NEXT_PUBLIC_* value +docker compose --env-file docker.custom.env up -d --build frontend ``` --- ## 10. Success Checklist -Your Docker deployment is successful when: - -- โœ… All containers are running: `docker compose ps` -- โœ… Ollama is accessible: `curl http://localhost:11434/api/tags` +- โœ… All containers healthy: `docker compose ps` +- โœ… Ollama reachable: `curl http://localhost:11434/api/tags` - โœ… Frontend loads: `curl http://localhost:3000` - โœ… Backend responds: `curl http://localhost:8000/health` -- โœ… RAG API works: `curl http://localhost:8001/models` -- โœ… You can create indexes and chat with documents - -### Performance Expectations - -**Acceptable Performance:** -- Container startup: < 2 minutes -- Memory usage: < 4GB Docker containers + Ollama -- Response time: < 30 seconds for complex queries - -**Optimal Performance:** -- Container startup: < 1 minute -- Memory usage: < 2GB Docker containers + Ollama -- Response time: < 10 seconds for complex queries +- โœ… RAG API responds: `curl http://localhost:8001/health` +- โœ… You can create an index and chat with your documents + +### What to Expect + +- **First `up --build`** is slow: it installs the Python dependencies (torch, + transformers, docling) and builds the Next.js bundle, and `rag-api` then + downloads the ~1.2GB embedding model before it reports healthy (the ~7.5GB + reranker downloads lazily, on the first reranked query) +- **Restarting** an existing container is fast. The HuggingFace weights are + downloaded at runtime into the container's writable layer, and no volume is + mounted for them, so **recreating** `rag-api` (any `up --build`, `down` + `up`, or + image change) downloads them again. Mount a cache directory and set `HF_HOME` if + that matters to you. +- **One RAG request at a time** โ€” the RAG API is single-threaded --- -**Happy Containerizing! ๐Ÿณ** \ No newline at end of file +**Happy Containerizing! ๐Ÿณ** diff --git a/Documentation/improvement_plan.md b/Documentation/improvement_plan.md index 1c84c5df..6f03e249 100644 --- a/Documentation/improvement_plan.md +++ b/Documentation/improvement_plan.md @@ -1,87 +1,142 @@ -# RAG System โ€“ Improvement Road-map +# localGPT โ€” Improvement Road-map -_Revision: 2025-07-05_ +_Revision: 2026-08-09_ -This document captures high-impact enhancements identified during the July 2025 code-review. Items are grouped by theme and include a short rationale plus suggested implementation notes. **No code has been changed โ€“ this file is planning only.** +Planned work only. Nothing in the **Open** sections is implemented. The **Landed** section records changes that were verified in the current working tree at the revision date โ€” every entry there names the file that proves it, so the list can be re-checked rather than trusted. + +> An evidence-based, phased extension of this plan lives in +> [research_roadmap.md](research_roadmap.md), grounded in the August 2026 +> research sweeps under [research/](research/). Items graduate from there into +> this file's Landed table as they ship. + +--- + +## 0. Landed (verified in-tree, 2026-08-08; Phase 1 rows added 2026-08-09) + +| Area | Change | Verify at | +|------|--------|-----------| +| Architecture | One RAG API server; the parallel `api_server_with_progress.py` is gone | `rag_system/` has a single `api_server.py` | +| Architecture | `factory.py` is the only agent/pipeline factory; `main.py` is config + a thin `index` / `chat` / `api` CLI | `rag_system/factory.py`, `rag_system/main.py` | +| Architecture | Backend gateway is threaded and every RAG API call has a timeout (`RAG_API_TIMEOUT`, `RAG_API_INDEX_TIMEOUT`) | `backend/server.py` | +| Ops | RAG API exposes `GET /health`; `run_system.py --health`, `--stop`, `--logs-only` and `--mode prod` all work | `rag_system/api_server.py`, `run_system.py` | +| Config | Service URLs and model ids come from environment variables, documented in `.env.example` | `.env.example`, `rag_system/main.py`, `backend/server.py`, `src/lib/api.ts` | +| Retrieval | Hybrid search fuses the full-text and vector legs with reciprocal rank fusion instead of a broken weighted blend; `retrieval_mode` (`hybrid` / `vector_only` / `fts_only`) is honoured | `rag_system/retrieval/retrievers.py` | +| Retrieval | **1.1 Late-chunk result merging** โ€” retrieved late-chunks are merged with their ยฑ1 siblings before reranking | `rag_system/pipelines/retrieval_pipeline.py` | +| Retrieval | A reranker that fails to load logs a warning and is skipped instead of throwing | `rag_system/pipelines/retrieval_pipeline.py::_get_ai_reranker` | +| Indexing | **3.3 Auto GPU dtype selection** โ€” CUDA > MPS > CPU with fp16 off CPU, for both the embedder and the late-chunk encoder | `rag_system/indexing/representations.py`, `rag_system/indexing/latechunk.py` | +| Indexing | Vector width is derived from the loaded model, and appending mismatched vectors to an existing table raises with a "rebuild the index" message (part of **3.4**) | `rag_system/indexing/embedders.py::VectorIndexer` | +| Indexing | OCR engine is chosen by probing which backend is installed, so a missing macOS-only engine no longer breaks every conversion | `rag_system/ingestion/document_converter.py` | +| Storage | The LanceDB path is resolved from `LANCEDB_PATH` / the pipeline config instead of three hard-coded literals, so index deletion targets the store indexing writes to | `backend/database.py::resolve_lancedb_path` | +| Privacy | The semantic cache defaults to `cache_scope: "session"`, closing the cross-session answer leak | `rag_system/main.py`, `rag_system/agent/loop.py` | +| Hygiene | DSPy modules and the non-functional ReAct agent are gone from the code; `rank_bm25`, `scikit-learn`, `nltk`, `sentence_transformers`, `colpali-engine` and `matplotlib` were dropped from `requirements.txt` | `grep` returns no code hits for any of them | +| Retrieval | **Phase 1.2 embedder** (`research_roadmap.md` ยง1.2) โ€” default embedder is `microsoft/harrier-oss-v1-0.6b` (MIT, 1024-dim) with the query-side `Instruct: โ€ฆ Query: โ€ฆ` prefix on for instruction-tuned families; documents stay unprefixed. Measured mixed-corpus first-stage nDCG@10 0.915 vs 0.875 for the previous `Qwen3-Embedding-4B` default, at ~3ร— lower latency (0.911 for the shipped stack once vectors are normalized โ€” see the row below). `EMBEDDING_MODEL` still overrides | `rag_system/main.py::EXTERNAL_MODELS`, `rag_system/indexing/representations.py::default_query_instruction`, `rag_system/pipelines/retrieval_pipeline.py::_query_instruction`, `eval/DECISIONS.md` | +| Retrieval | **Phase 1.1 reranker** (`research_roadmap.md` ยง1.1) โ€” the `default` profile shipped `reranker.enabled = False`: with this first stage the cheap cross-encoder measures net-negative (0.915 โ†’ 0.892 mixed) and the reranker that wins costs ~12.7 s/query. **Superseded 2026-08-14 (arm G, see ยง1.9): reranking is now ON by default with `min_score: 0.5` / `min_keep: 3` threshold selection.** `Qwen/Qwen3-Reranker-4B` loads lazily on the first reranked query through the in-repo `QwenRerankerScorer` | `rag_system/main.py::PIPELINE_CONFIGS`, `rag_system/pipelines/retrieval_pipeline.py::_get_ai_reranker`, `src/components/ui/session-chat.tsx`, `eval/DECISIONS.md` | +| Indexing | **Embedder-identity guard** โ€” the width-only check cannot catch a swap between two 1024-dim models, so every table records the embedding model that wrote it (Arrow schema metadata, sidecar fallback). Indexing into or querying a table with a different embedder raises `EmbedderMismatchError` instead of returning nonsense | `rag_system/indexing/embedders.py::read_table_marker`/`assert_embedder_matches`, `rag_system/retrieval/retrievers.py::MultiVectorRetriever._check_table_identity` | +| Indexing | **Cosine normalization** โ€” vectors are L2-normalized at write and query time, so LanceDB's default L2 ordering is the cosine ordering both model cards specify. Gated per table on the `normalized` marker, so legacy tables keep working unnormalized with a warning; the default table moved to `text_pages_v4`. Measured a wash on the gold set (mixed nDCG@10 0.915 โ†’ 0.911, recall@5 0.917 โ†’ 0.931): adopted for card-conformance, not for a number | `rag_system/indexing/embedders.py::l2_normalize`, `rag_system/retrieval/retrievers.py`, `rag_system/main.py` | +| Evaluation | **Phase 0 harness** (`research_roadmap.md` ยง0) โ€” three corpora (two planted-fact PDFs + `Documentation/*.md`), a 72-query gold set labelled by answer-bearing text rather than chunk id, an in-process recall@5/10/20 + nDCG@10 runner, a binary groundedness judge validated at TPR 1.00 / TNR 1.00 on 20 hand-labelled cases, and a scripted end-to-end smoke that passes 25/25 assertions. Baseline recorded; every Phase 1/2 item now has something to A/B against | `eval/run_eval.py`, `eval/judge.py`, `eval/smoke_e2e.py`, `eval/goldset/*.jsonl`, `eval/BASELINE.md` | +| Models | **Roadmap 1.1/1.2 โ€” evidence-gated defaults**: embedder `microsoft/harrier-oss-v1-0.6b` + query instruction prefix (mixed nDCG@10 0.915 vs 0.875 for the 8 GB Qwen3-4B); default profile reranker **off** (bge measured net-negative on this first stage; Qwen3-Reranker-4B is the lazy opt-in at +0.06 nDCG for ~12.7 s/query) โ€” **superseded 2026-08-14 (arm G): reranker ON by default with `min_score`/`min_keep` selection**; per-table embedder-identity markers + L2 normalization, `text_pages_v4` | `rag_system/main.py`, `rag_system/indexing/embedders.py`, `eval/DECISIONS.md` | +| Routing | **Roadmap 2.3 โ€” gateway routing is a deterministic gate.** The per-message enrichment-model router and the `_simple_pattern_routing` keyword/length fallback are deleted; `should_use_rag()` routes on `force_rag` โ†’ linked indexes โ†’ a whole-message smalltalk/assistant-meta allowlist โ†’ RAG. ~750 ms/message saved; agent triage is now the only LLM routing layer | `backend/server.py::should_use_rag`, `backend/test_gateway_routing.py` (155/155), `eval/decisions/phase2-gateway.md` | +| Retrieval | **Roadmap 2.5 โ€” graph module removed**: `GraphExtractor`, `GraphRetriever`, `GraphQueryTranslator`, the `graph_query` triage outcome and the `retrieval.graph` / `graph_strategy` config keys are gone; `networkx`, `fuzzywuzzy`, `python-Levenshtein` dropped from all requirements files. Contested gains, 41โ€“57ร— indexing and up to ~377ร— query-token cost (`research/academic-evidence-2026.md` ยง6) | `rag_system/indexing/` has no `graph_extractor.py`; `eval/decisions/phase2-pipeline.md` ยง1 | +| Retrieval | **Roadmap 2.1 โ€” evidence-sufficiency retry**: one conditional second retrieval on weak evidence (candidate-set contrast signal โ€” raw top similarity measured anti-correlated), on in `default`, off in `fast`. Fires on ~10% of queries, +0.008โ€“0.017 nDCG@10, zero per-query regressions in four runs | `rag_system/pipelines/retrieval_pipeline.py::retrieve_candidates`, `eval/decisions/phase2-pipeline.md` ยง2 | +| Retrieval | **Roadmap 2.2 โ€” decomposition at rerank**: the first stage always runs once on the full query; sub-queries score candidates at rerank (`query_decomposition.rerank_aggregate`), and first-stage fan-out survives only behind `compose_from_sub_answers`. Measured negative on truly-decomposing queries โ€” no shipped profile enables rerank-decomposition. **Superseded in part: arm G (2026-08-14) turned the reranker on by default, so rerank-decomposition now applies; arm H (2026-08-15) made the pooled first stage the default (`pooled_first_stage: true`, `compose_from_sub_answers: false`) โ€” per-sub-query retrieval, pooled + deduped candidates, ONE rerank + ONE synthesis** | `rag_system/pipelines/retrieval_pipeline.py`, `eval/decisions/phase2-pipeline.md` ยง3 | +| Verification | **Roadmap 2.4 โ€” verifier model seam**: `VERIFIER_MODEL` / `verification.model` swaps the LLM-prompt verifier for a local NLI model (MiniCheck-DeBERTa 19/20, DeBERTa-MNLI 18/20 on the judge set); default unchanged. ThinknCheck has no public weights | `rag_system/agent/verifier.py::LocalNLIVerifier`, `eval/decisions/phase2-pipeline.md` ยง4 | +| Eval | `run_eval.py` drives `RetrievalPipeline.retrieve_candidates()` โ€” first stage, retry and rerank are the shipped code path; `--retry`, `--decompose`, `--aggregate` flags added; gold rows `docs_d10` + the triage-model row re-anchored after 2.5/2.3 deleted their source text (recorded per-row) | `eval/run_eval.py`, `eval/goldset/docs.jsonl` | +| Observability | **Roadmap 4.5 โ€” per-query token accounting** (on by default): Ollama `prompt_eval_count`/`eval_count` aggregated per user query, bucketed by stage, returned as `token_usage` on `/chat` and the SSE `complete` event. ContextVar-based; watsonx reports zeros; embedding calls uncounted | `rag_system/utils/ollama_client.py::TokenUsageTracker`, `eval/decisions/phase4-escalation-tokens.md` | +| Retrieval | **Roadmap 4.4 โ€” metadata filter DSL** (no flag; inert without a `filters` argument, md5-verified byte-identical when unused): JSON filters compiled to LanceDB where-clauses prefiltering both search legs; quoting characters refused, malformed filters are the RAG API's 400; gateway forwards `filters` and treats a present filter as force-RAG. Page/date filtering NOT shipped (needs real columns + re-index โ†’ ยง3.7) | `rag_system/retrieval/filters.py`, `rag_system/api_server.py`, `backend/server.py`, `eval/decisions/phase4-filters-askfolder.md` | +| CLI | **Roadmap 4.6 โ€” ask-a-folder**: `python -m rag_system.main ask ""` builds an ephemeral index (fast profile), answers with the standard pipeline, cleans up everything including on SIGTERM | `rag_system/ask_folder.py`, `eval/decisions/phase4-filters-askfolder.md` | +| Indexing | **Roadmap 4.2 (index half) โ€” cross-reference extraction** (on by default): regex-only extraction of exhibit/section/document-name references into `metadata.crossrefs`, filename resolution incl. numeric-prefix-stripped aliases; verified bit-identical text/vector columns. The **query-time hop is REJECTED as a default** (flag kept, off): 0/11 fires at the shipped k=20, 0/11 target precision where forced, beaten by raising `k` at equal budget | `rag_system/indexing/crossref.py`, `eval/decisions/phase4-retrieval-benchmarks.md`, `eval/decisions/phase4-answer-quality.md` | +| Retrieval | **Roadmap 4.1 escalation โ€” REJECTED as default (2026-08-12 re-run); 4.3 overview prefilter โ€” HELD off**: the HOLD's re-run condition was executed after the num_ctx fix and the lift did not survive โ€” the escalation-off baseline on the identical fire subset went 0/9โ†’7/9 (the 2026-08-09 "lift" was front-truncation favoring the tail-appended document) and both product-default fires were regressions (`eval/decisions/phase4-escalation-rerun.md`); prefilter `boost` wins only on heterogeneous corpora, `restrict` is rejected outright (kills recall@20 on 4 queries/corpus). Both stay flag-gated off | `rag_system/agent/escalation.py`, `rag_system/pipelines/retrieval_pipeline.py`, `eval/decisions/phase4-*.md`, `design_rationale.md` ยง13a | + +The previous revision of this file marked five cleanup items โœ… COMPLETED. Two of those claims (removing unused imports/dependencies, consolidating configuration files) were not true when written and are only partly true now โ€” the surviving work is tracked in ยง9 below. --- -## 1. Retrieval Accuracy & Speed +> **Chunker data loss โ€” FIXED 2026-08-12, REBUILD INDEXES.** `MarkdownRecursiveChunker._split_text`'s separator-reassembly loop dropped the first and every other body segment of any document large enough to split, so long documents entered the index at ~50% of their content (RFC 9000: 49% retained; `design_rationale.md`: 49%). Found by the unseen-corpus RFC shakedown โ€” the authored eval corpora were too small to trigger the path. Fixed and property-tested lossless (30 randomized structured docs, zero non-whitespace loss; post-fix retention 94โ€“99.8%, remainder is whitespace normalization). **Every index built before this fix under-contains its long documents** โ€” product indexes in `index_store/` and the tracked eval indexes should be rebuilt, and pre-fix retrieval baselines on the `docs` corpus are not comparable to post-fix numbers. + +## 1. Retrieval accuracy & speed | ID | Item | Rationale | Notes | |----|------|-----------|-------| -| 1.1 | Late-chunk result merging | Returned snippets can be single late-chunks โ†’ fragmented. | After retrieval, gather sibling chunks (ยฑ1) and concatenate before reranking / display. | -| 1.2 | Tiered retrieval (ANN pre-filter) | Large indexes โ†’ LanceDB full scan can be slow. | Use in-memory FAISS/HNSW to narrow to top-N, then exact LanceDB search. | -| 1.3 | Dynamic fusion weights | Different corpora favour dense vs BM25 differently. | Learn weight on small validation set; store in index `metadata`. | -| 1.4 | Query expansion via KG | Use extracted entities to enrich queries. | Requires Graph-RAG path clean-up first. | +| 1.2 | Tiered retrieval (ANN pre-filter) | Large tables make LanceDB scans slow. | Narrow to top-N with an in-memory index, then exact search. | +| 1.3 | Corpus-tuned fusion | RRF is weight-free and safe, but a tuned fusion could beat it per corpus. | Would need a validation set and a place to store the setting per index. | +| 1.4 | Query expansion via extracted entities | Richer queries for entity-heavy corpora. | Depends on the Graph-RAG path (ยง9) being finished or removed. | +| 1.5 | ~~Deduplicate the late-chunk leg~~ **FIXED 2026-08-14** | Late-chunk hits were appended to the base hits with no dedupe, so the same passage occupied two slots โ€” and with the ยฑ1 sibling merge tripling every entry, synthesis context reached ~94k tokens/call while Ollama's slot-split window served only ~16k and silently front-truncated (the model saw the *tail* of the ranking). | `retrieval_pipeline.py` now dedupes across both legs on `(document_id, chunk_index)` and packs rank-ordered docs into an explicit synthesis token budget (default 12k, `synthesis_context_tokens` config) with sibling-span overlap suppression. Measured (arm F, `eval/decisions/synthesis-grounding-ab-2026-08-13.md`): context 335k โ†’ ~40k chars, truncation warnings 32 โ†’ 0, E2E wall time โˆ’32%, judged pass **7/24 โ†’ 16/24**. | +| 1.6 | Query-aware crossref-hop targets | The 4.2 hop was rejected as a default partly because target selection is query-blind: `max_hops=1` takes the first unrepresented reference in scan order, which lands on hub documents (21/24 hops โ†’ 2 docs, 0/11 expected-source precision). | Score candidate targets against the query (overview embeddings exist) before spending the hop; then re-run `eval/decisions/phase4-retrieval-benchmarks.md`'s A/B. | +| 1.8 | Crossref extraction for real-world naming conventions | The unseen-corpus RFC shakedown measured the extractor at **0/1403 resolved** on 23 real RFCs: filenames never occur in the prose, 355 explicit `[QUIC-TRANSPORT] Section N` cross-document references are discarded by `_SECTION_RE`, and bare `RFC NNNN` mentions have no pattern (`eval/decisions/rfc-shakedown-2026-08-13.md`). The extractor's patterns encode the authored acq corpus's conventions. | Add bracketed-citation and `RFC NNNN`-style patterns + title-based (not just filename-based) resolution; re-measure on the rfc corpus (a rename experiment showed 91 resolutions are reachable). Index-inert either way; the hop stays off. | +| 1.9 | Synthesis grounding on dense unseen technical text | With retrieval fixed (recall@20 0.958 on the rfc corpus), end-to-end answer quality is the measured bottleneck: **5/24** judged pass (Sonnet panel; single-doc 1/14, crossref 4/10). The 9b answers from its prior instead of the supplied snippets and fabricates citations (e.g. inventing a quote from "RFC 9002 ยง13.4"). The verifier flags these low-confidence but the wrong claim still leads the answer. | A/B run 2026-08-13 (`eval/decisions/synthesis-grounding-ab-2026-08-13.md`): strict snippet-only prompt + temperature-0 decode ADOPTED as default (7/24 vs 5/24 โ€” at the noise floor, adopted for escape-hatch removal + determinism, not claimed as a quality win); **qwen3.6:35b-a3b REJECTED** (equal-or-worse at 4-5/24 under both prompts โ€” grounding is not parameter count at this scale). Strict compose prompt TESTED AND REJECTED same day (arm E: 4/24 vs C's 7/24; all three passโ†’fail flips were composed rows โ€” strict copy-rules against already-synthesized sub-answer prose drop facts the sub-answers carried; reverted, decision file appendix). **2026-08-14 (arm F): the dominant cause was context overflow, not the model** โ€” synthesis prompts were ~94k tokens against a slot-split ~16k served window, so front-truncation deleted the top-ranked evidence on every call. Cross-leg dedupe + 12k context budget (item 1.5) lifted judged pass to **16/24 (single-doc 10/14, crossref 6/10)** with a near-unanimous panel (1 split/72 votes) and โˆ’32% wall time. Item narrows to the crossref/multi-hop residue: next levers are deterministic decomposition (temp 0 โ€” also stabilizes A/B row sets); abstain-on-low-verifier-confidence; passage-level citation forcing. **2026-08-15 (arm G): Qwen3-Reranker-4B enabled by default as final-stage SELECTION** โ€” union-of-max `min_score 0.5` over original+sub-queries keeps only candidates the calibrated scorer marks relevant (mean 8.8, range 3โ€“10 โ€” adaptive context), scored on preserved core chunk text; **18/24 (11/14, 7/10)** vs arm F 16/24 โ€” net +2 within noise floor, adopted as user-directed no-regression feature at +92% wall time (opt-out `reranker.enabled: false`; 0.6B scorer unmeasured latency lever). Supersedes the Phase-1 off-by-default call, which predated the context budget. **2026-08-15 (arms H/H2/G2): decomposition restructured to POOLED** โ€” per-sub-query retrieval pooled+deduped, one source-aware rerank (per-SQ floor guarantees every sub-question keeps evidence), one synthesis with the original query; composer eliminated from the path. Decomposition now decodes at **temperature 0** (deterministic splits โ€” A/B row sets finally stable; all 18 direct rows got identical panel verdicts across the H2/G2 re-run). Clean comparison on identical row sets: pooled 17/24 vs composer 17/24 โ€” true tie; pooled adopted on structure (24 vs 32 synthesis calls, linear scaling in sub-query count, no compose-step fact loss). **2026-08-15 (arm I): source-document labels โ€” 21/24 (single-doc 12/14, crossref 9/10), +4/0 vs H2, first clearly-attributable synthesis quality win.** Diagnosis showed half the crossref failures were attribution failures (answers with every fact correct, failed for not naming "RFC 9001"/"RFC 9000"): the context join carried no source labels while prompt rule 3 forbids doc names not in snippets. Every snippet now opens with `[Source document: ]` + rule 7 permits attribution via labels. Remaining residue: rfc_q10/q15 (single-doc), rfc_q17 (nuance), rfc_q20 (gold-nuance split). | +| 1.7 | ~~Sanitize FTS sub-query input~~ **FIXED 2026-08-12** | A decomposer-emitted sub-query containing double quotes made the LanceDB FTS parser raise ("position is not found but required for phrase queries"), killing hybrid retrieval for that query outright (1/24 gold queries during the Phase-4 A/B). | `retrievers.py` now strips double quotes before the FTS leg, and a hybrid search whose FTS leg fails for any other reason degrades to dense-only with a warning instead of returning nothing (`fts_only` mode still propagates). Verified against the exact incident query: 0 docs โ†’ 5 docs, answer document ranked first; degradation tested with a simulated FTS failure. | -## 2. Routing / Triage +## 2. Routing / triage | ID | Item | Rationale | |----|------|-----------| -| 2.1 | Embed + cache document overviews | LLM router costs tokens; cosine-similarity pre-check is cheaper. | -| 2.2 | Session-level routing memo | Avoid repeated LLM triage for follow-up queries. | -| 2.3 | Remove legacy pattern rules | Simplifies maintenance once overview & ML routing mature. | +| 2.1 | Embed and cache document overviews | The agent router (now the only LLM routing layer) makes an LLM call per query; a cosine pre-check would be far cheaper. | +| 2.2 | Session-level routing memo | Today the only shortcut is "history exists โ†’ `rag_query`". Cache the decision instead. | -## 3. Indexing Pipeline +## 3. Indexing pipeline | ID | Item | Rationale | |----|------|-----------| -| 3.1 | Parallel document conversion | PDFโ†’MD + chunking is serial today; speed gains possible. | -| 3.2 | Incremental indexing | Re-embedding whole corpus wastes time. | -| 3.3 | Auto GPU dtype selection | Use FP16 on CUDA / MPS for memory and speed. | -| 3.4 | Post-build health check | Catch broken indexes (dim mismatch etc.) early. | +| 3.1 | Parallel document conversion | Conversion and chunking are serial per file. | +| 3.2 | Incremental indexing | Re-embedding the whole corpus to add one document is wasteful. | +| 3.4 | Post-build health check | The dimension guard exists; a build should also assert the FTS index, the row count and the late-chunk table when requested. | +| 3.5 | Align chunk-size defaults | No profile sets `chunking.chunk_size`, so CLI indexing chunks at 1500 tokens while `POST /index` defaults to 512 (`api_server.py:441`) โ€” same class of split as 3.6. Documented in `system_overview.md` ยง5.3. | +| 3.6 | Align late-chunk defaults | `POST /index` defaults `enable_latechunk` to `false` while the `default` profile enables late-chunk retrieval, so HTTP builds and CLI builds silently differ (retrieval degrades gracefully when the `_lc` table is absent). | +| 3.7 | Real `page`/`date` columns | Roadmap 4.4 promised page/date filters; they did not ship because both live inside the metadata JSON string column (`metadata LIKE` would false-match `"page": 31` for page 3). Needs first-class columns + a re-index window. | -## 4. Embedding Model Management +## 4. Model management -* **Registry file** mapping tag โ†’ dims/source/license. UI & backend validate against it. -* **Embedder pool** caches loaded HF/Ollama weights per model to save RAM. +* Registry mapping model tag โ†’ dimensions, source and license, validated by the UI and both servers. +* An embedder pool that keeps one copy of each loaded model in memory (a module-level cache exists in `representations.py`; the reranker and pruner have their own ad-hoc singletons). +* Warn โ€” or refuse โ€” when the embedding model configured for a query differs from the one recorded in the index metadata. -## 5. Database & Storage +## 5. Database & storage -* LanceDB table GC for orphaned tables. -* Scheduled SQLite `VACUUM` when fragmentation > X %. +* Garbage-collect orphaned LanceDB tables (an index row deleted outside the API leaves its table behind). +* Delete files from `shared_uploads/` when their session or index is deleted. +* Scheduled SQLite `VACUUM` when fragmentation is high. -## 6. Observability & Ops +## 6. Observability & ops -* JSON structured logging. -* `/metrics` endpoint for Prometheus. -* Deep health-probe (`/health/deep`) exercising end-to-end query. +* JSON structured logging and log rotation in `run_system.setup_logging` (today: plain text, no rotation). +* Move the agent's and pipelines' `print()` progress output onto the `logging` module. +* A `/metrics` endpoint for Prometheus. +* A deep health probe (`/health/deep`) that runs a real end-to-end query. +* Per-request pipeline configuration on the RAG API, so options stop leaking between requests, followed by switching the RAG API to a threading server. *(The leak half has landed: the agent now snapshots its pipeline config before applying per-request overrides and restores it afterwards โ€” `rag_system/agent/loop.py`. The threading-server switch remains.)* +* **Context-window budgeting โ€” FIXED 2026-08-12 (found by the Phase-4 A/B, 2026-08-09):** `ollama_client` never set `num_ctx`, and the served ceiling on this host was **8194 prompt tokens** with silent *front*-truncation โ€” a 373-chunk corpus's synthesis prompt (20โ€“25k tokens) lost its top-ranked evidence without any error. Now both clients (`rag_system/utils/ollama_client.py`, `backend/ollama_client.py`) size `options.num_ctx` per request from the prompt length, bucketed 8k/16k/32k and per-model **monotonic** โ€” the window only ratchets upward, because changing `num_ctx` between calls forces a KV-cache reallocation (measured on the smoke suite: naive per-call bucketing 924s; fixed 32k window 259s; the shipped monotonic ratchet 208s; pre-fix baseline 363s). `OLLAMA_NUM_CTX` pins a value, `OLLAMA_NUM_CTX_MAX` caps the bucket (default 32768), and a filled window (`prompt_eval_count โ‰ฅ num_ctx โˆ’ 16`) logs a loud truncation warning. Verified 2026-08-12: the 100k-char probe that returned `prompt_eval_count: 8194` now evaluates the full prompt (17,351 tokens) and recovers a fact planted at position 0. The 4.1 escalation A/B re-run this unblocked was executed 2026-08-12 (`eval/decisions/phase4-escalation-rerun.md`; verdict: reject). Note the fix also **voids** the 2026-08-09 `acqdocs` baseline in `phase4-answer-quality.md` ยง4 โ€” that 0/9 was a truncated-context artifact and reads 7/9 on identical inputs post-fix. Still open: prompts >~90k chars hit the 32k default cap โ€” client-side facts-string budgeting remains worthwhile. Related, **FIXED 2026-08-12**: the synthesis stream paths never set `think:false`, so the 9b could burn its window on chain-of-thought and return an empty answer (measured: prompt 9,351 + thinking 7,033 = window exactly, empty `response`). All three streaming synthesis call sites (`retrieval_pipeline._synthesize_final_answer` and both agent-loop compose paths) now pass `enable_thinking=False`; verified at the wire that `think: false` reaches the Ollama payload. The eval harness's `patch_no_think` monkeypatch is now redundant for these paths but harmless. +* Render `token_usage` in the UI (it reaches the browser on the SSE `complete` event and is persisted in the turn's steps snapshot; `conversation-page.tsx` doesn't display it yet), and forward it on the gateway's non-streaming path (`backend/server.py::_handle_rag_query` extracts only `answer` + `source_documents`). +* Judge noise is larger than the Phase-0 validation suggested: 5โ€“7 verdict flips per 24 rows between identical runs were observed in the Phase-4 A/Bs. Direction-deciding cells need kโ‰ฅ5 judge votes (`eval/decisions/phase4-answer-quality.md`). The smoke test shares the underlying cause โ€” its substring assertions run against temperature-1.0 generations, and 1 of 6 runs on 2026-08-09 flaked on one question (a rerun passed 25/25). Consider `temperature 0` for smoke/eval generation calls. Worse, the 2026-08-12 re-run caught the judge returning verdicts its own reasons contradict (an answer containing the gold fact verbatim voted 0/5), and being perturbed by the verifier's `[Confidence: N%] [Warning: โ€ฆ]` suffix, which one judge reason cited as grounds for rejection. Two mitigations landed 2026-08-12: `eval/judge.py` now strips the verifier suffix from both judge slots before scoring, and the judge gained an Anthropic-API backend (eval-only; `JUDGE_MODEL=claude-sonnet-5` routes any `claude-*` model name through the `anthropic` SDK with a server-enforced JSON schema โ€” the product stays fully local). The 18 hand-adjudicated rows from the re-run are preserved as a permanent hard-case benchmark (`eval/judge_hard_cases.jsonl` + `eval/validate_judge_hard.py`; the 4b's k=5 majority scores 13/18 on it โ€” any candidate judge must beat that). **A Sonnet-class judge is now validated (2026-08-13)**: three independent Claude Sonnet subagent voters, each running the exact v1 prompt (suffix-stripped) over all 38 cases, scored **20/20 on the Phase-0 validation set and 18/18 on the hard set, with zero split votes across 114 judgments** โ€” including the fact-present-but-prefaced-with-a-denial row both 4b arms failed (`eval/decisions/judge-sonnet-validation-2026-08-13.json`). Protocol for direction-deciding cells: generate the per-case v1 prompts (as `eval/judge.py` builds them) and fan them out to 3 Sonnet subagent voters, majority decides; the 4b remains the free bulk-pass judge. The `JUDGE_MODEL=claude-*` API backend in `eval/judge.py` runs the identical prompt and is available when API credentials exist. +* **Whole-document context can send the 9b model into verbatim transcription**: with escalation forced on and an untruncated 32k window, one query produced a 35,290-char answer (vs 1,400 baseline) that answered correctly in its first paragraph and then regurgitated `design_rationale.md` text wholesale (`eval/decisions/phase4-escalation-rerun.md` ยง8.1). Any future feature that injects large verbatim blocks needs an output cap or a "do not transcribe the context" instruction in the synthesis prompt. ## 7. Front-end UX -* SSE-driven progress bar for indexing. +* SSE-driven progress for indexing (chat already streams phases; indexing is a blocking POST). * Matched-term highlighting in retrieved snippets. -* Preset buttons (Fast / Balanced / High-Recall) for retrieval settings. +* Preset buttons (Fast / Balanced / High-Recall) over the retrieval settings. +* Surface the verifier's confidence as a field rather than parsing it out of the answer string. ## 8. Testing & CI -* Replace deleted BM25 tests with LanceDB hybrid tests. -* Integration test: build โ†’ query โ†’ assert โ‰ฅ1 doc. -* GitHub Action that spins up Ollama, pulls small embedding model, runs smoke test. - -## 9. Codebase Hygiene +There is no automated test suite. The only checks are `python system_health_check.py`, `python run_system.py --health` and `./test_docker_build.sh`. -* Graph-RAG integration (currently disabled, can be implemented if needed). -* Consolidate duplicate config keys (`embedding_model_name`, etc.). -* Run `mypy --strict`, pylint, and black in CI. +* Unit tests for `MultiVectorRetriever.retrieve` across all three modes, including the RRF ordering. +* Integration test: build an index โ†’ query โ†’ assert at least one source document. +* A GitHub Action that starts Ollama, pulls a small embedding model and runs the smoke test. +* Re-enable the type/lint gates that `next.config.ts` currently disables (`eslint.ignoreDuringBuilds`, `typescript.ignoreBuildErrors`). ---- +## 9. Codebase hygiene -### ๐Ÿงน System Cleanup (Priority: **HIGH**) -Reduce complexity and improve maintainability. +* `docker-compose.local-ollama.yml` now duplicates `docker-compose.yml` (which already defaults to host Ollama). Pick one. *(Resolved: the variant file was removed โ€” `docker-compose.yml` plus `--profile with-ollama` is the only flow.)* +* Run `mypy`, `pylint` and `black` in CI. -* **โœ… COMPLETED**: Remove experimental DSPy integration and unused modules (35+ files removed) -* **โœ… COMPLETED**: Clean up duplicate or obsolete documentation files -* **โœ… COMPLETED**: Remove unused import statements and dependencies -* **โœ… COMPLETED**: Consolidate similar configuration files -* **โœ… COMPLETED**: Remove broken or non-functional ReAct agent implementation +--- -### Priority Matrix (suggested order) +### Priority matrix (suggested order) -1. **Critical reliability**: 3.4, 5.1, 9.2 -2. **User-visible wins**: 1.1, 7.1, 7.2 -3. **Performance**: 1.2, 3.1, 3.3 -4. **Long-term maintainability**: 2.3, 9.1, 9.3 +1. **Correctness / data loss**: 3.6 +2. **User-visible wins**: 7.1, 7.2, 2.4 +3. **Reliability**: 3.4, 5.1, 8.2 +4. **Performance**: 1.2, 1.5, 3.1 +5. **Long-term maintainability**: 2.1, 4, 9 -Feel free to rearrange based on team objectives and resource availability. \ No newline at end of file +Rearrange to suit team objectives and available time. diff --git a/Documentation/indexing_pipeline.md b/Documentation/indexing_pipeline.md index 009ee497..7d932c68 100644 --- a/Documentation/indexing_pipeline.md +++ b/Documentation/indexing_pipeline.md @@ -1,665 +1,299 @@ # ๐Ÿ—‚๏ธ Indexing Pipeline -_Implementation entry-point: `rag_system/pipelines/indexing_pipeline.py` + helpers in `indexing/` & `ingestion/`._ +_Implementation entry-point: `rag_system/pipelines/indexing_pipeline.py`, with helpers in `rag_system/ingestion/` and `rag_system/indexing/`._ ## Overview -Transforms raw documents (PDF, TXT, etc.) into search-ready **chunks** with embeddings, storing them in LanceDB and generating auxiliary assets (overviews, context summaries). +Turns documents (PDF, DOCX, HTML/HTM, MD, TXT) into search-ready chunks with embeddings, writes them to LanceDB, builds a full-text index over the same table, and generates a per-document overview used by the triage routers. + +Public entry point: + +```python +IndexingPipeline.run(file_paths: list[str]) -> None +``` + +`run()` also accepts a legacy `documents=` keyword as an alias for `file_paths` (`indexing_pipeline.py:157-167`). It returns `None`; progress and results are printed. + +## High-level diagram -## High-Level Diagram ```mermaid flowchart TD - A["Uploaded Files"] --> B{Converter} - B -->|PDFโ†’text| C["Plain Text"] - C --> D{Chunker} - D -->|docling| D1[DocLing Chunking] - D -->|latechunk| D2[Late Chunking] - D -->|standard| D3[Fixed-size] - D1 & D2 & D3 --> E["Contextual Enricher"] - E -->|local ctx summary| F["Embedding Generator"] - F -->|vectors| G[(LanceDB Table)] - E --> H["Overview Builder"] - H -->|JSONL| OVR[[`index_store/overviews/.jsonl`]] + A["Files"] --> B["DocumentConverter (docling)"] + B --> C["Markdown + DoclingDocument"] + C --> D{"chunker_mode"} + D -- docling --> D1["DoclingChunker.chunk_document()"] + D -- legacy --> D2["MarkdownRecursiveChunker.chunk()"] + D1 --> OV["OverviewBuilder (per document)"] + D2 --> OV + OV --> OVF[["index_store/overviews/<id>.jsonl"]] + D1 --> E["ContextualEnricher (optional)"] + D2 --> E + E --> F["EmbeddingGenerator"] + F --> G[("LanceDB table")] + G --> H["create_fts_index('text')"] + F --> LC["LateChunkEncoder (optional)"] + LC --> GLC[("LanceDB table <table>_lc")] ``` -## Steps in Detail -| Step | Module | Key Classes | Notes | -|------|--------|------------|-------| -| Conversion | `ingestion/pdf_converter.py` | `PDFConverter` | Uses `Docling` library to extract text with structure preservation. | -| Chunking | `ingestion/chunking.py`, `indexing/latechunk.py`, `ingestion/docling_chunker.py` | `MarkdownRecursiveChunker`, `DoclingChunker` | Controlled by flags `latechunk`, `doclingChunk`, `chunkSize`, `chunkOverlap`. | -| Contextual Enrichment | `indexing/contextualizer.py` | `ContextualEnricher` | Generates per-chunk summaries (LLM call). | -| Embedding | `indexing/embedders.py`, `indexing/representations.py` | `QwenEmbedder`, `EmbeddingGenerator` | Batch size tunable (`batchSizeEmbed`). Uses Qwen3-Embedding models. | -| LanceDB Ingest | `index_store/lancedb/โ€ฆ` | โ€“ | Each index has a dedicated table `text_pages_`. | -| Overview | `indexing/overview_builder.py` | `OverviewBuilder` | First-N chunks summarised for triage routing. | - -### Control Flow (Code) -1. **backend/server.py โ†’ handle_build_index()** collects files + opts and POSTs to `/index` endpoint on advanced RAG API (local process). -2. **indexing_pipeline.IndexingPipeline.run()** orchestrates conversion โ†’ chunking โ†’ enrichment โ†’ embedding โ†’ storage. -3. Metadata (chunk_size, models, etc.) stored in SQLite `indexes` table. - -## Configuration Flags -| Flag | Description | Default | -|------|-------------|---------| -| `latechunk` | Merge k adjacent sibling chunks at query time | false | -| `doclingChunk` | Use DocLing structural chunking | false | -| `chunkSize` / `chunkOverlap` | Standard fixed slicing | 512 / 64 | -| `enableEnrich` | Run contextual summaries | true | -| `embeddingModel` | Override embedder | `Qwen/Qwen3-Embedding-0.6B` | -| `overviewModel` | Model used in `OverviewBuilder` | `qwen3:0.6b` | -| `batchSizeEmbed / Enrich` | Batch sizes | 50 / 25 | - -## Error Handling -* Duplicate LanceDB table โžŸ now idempotent (commit `af99b38`). -* Failed PDF parse โžŸ chunker skips file, logs warning. - -## Extension Ideas -* Add OCR layer before PDF conversion. -* Store embeddings in Remote LanceDB instance (update URL in config). - -## Detailed Implementation Analysis - -### Pipeline Architecture Pattern -The `IndexingPipeline` uses a **sequential processing pattern** with parallel batch operations. Each stage processes all documents before moving to the next stage, enabling efficient memory usage and progress tracking. +## Steps in detail -```python -def run(self, file_paths: List[str]): - with timer("Complete Indexing Pipeline"): - # Stage 1: Document Processing & Chunking - all_chunks = [] - doc_chunks_map = {} - - # Stage 2: Contextual Enrichment (optional) - if self.contextual_enricher: - all_chunks = self.contextual_enricher.enrich_batch(all_chunks) - - # Stage 3: Dense Indexing (embedding + storage) - if self.vector_indexer: - self.vector_indexer.index_chunks(all_chunks, table_name) - - # Stage 4: Graph Extraction (optional) - if self.graph_extractor: - self.graph_extractor.extract_and_store(all_chunks) -``` +| Step | Module | Key classes | Notes | +|------|--------|-------------|-------| +| Conversion | `ingestion/document_converter.py` | `DocumentConverter` | docling for PDF/DOCX/HTML/MD; plain read for `.txt`. PyMuPDF (`fitz`) is used **only** to test whether a PDF already has a text layer. | +| Chunking | `ingestion/docling_chunker.py`, `ingestion/chunking.py` | `DoclingChunker`, `MarkdownRecursiveChunker` | Docling is the default. Both are token-aware via the embedding model's tokenizer. | +| Overview | `indexing/overview_builder.py` | `OverviewBuilder` | One LLM call per document, from its first N chunks. On by default. | +| Contextual enrichment | `indexing/contextualizer.py` | `ContextualEnricher` | One LLM call per chunk. On in the `default` profile. | +| Embedding | `indexing/representations.py` | `QwenEmbedder`, `OllamaEmbedder`, `EmbeddingGenerator` | `select_embedder()` picks HuggingFace vs Ollama. | +| LanceDB write | `indexing/embedders.py` | `LanceDBManager`, `VectorIndexer` | Appends to an existing table, creates it otherwise. Vector width is taken from the produced embeddings. | +| Full-text index | `pipelines/indexing_pipeline.py:264-284` | โ€“ | `tbl.create_fts_index("text", use_tantivy=False)`. This is what makes `hybrid` retrieval work. | +| Late chunking | `indexing/latechunk.py` | `LateChunkEncoder` | Optional second embedding pass into `
_lc`. | -### Document Processing Deep-Dive +Storage location is `storage.lancedb_uri` (also accepted as `storage.db_path` / `storage.lancedb_path`), which both profiles set to `./lancedb`. Per-index tables are named `text_pages_` (`backend/database.py:351`); the profile fallback table is `text_pages_v4`. -#### PDF Conversion Strategy -```python -# PDFConverter uses Docling for robust text extraction with structure -def convert_to_markdown(self, file_path: str) -> List[Tuple[str, Dict, Any]]: - # Quick heuristic: if PDF has text layer, skip OCR for speed - use_ocr = not self._pdf_has_text(file_path) - converter = self.converter_ocr if use_ocr else self.converter_no_ocr - - result = converter.convert(file_path) - markdown_content = result.document.export_to_markdown() - - metadata = {"source": file_path} - # Return DoclingDocument object for advanced chunkers - return [(markdown_content, metadata, result.document)] -``` +## Control flow -**Benefits**: -- Preserves document structure (headings, lists, tables) -- Automatic OCR fallback for image-based PDFs -- Maintains page-level metadata for source attribution -- Structured output supports advanced chunking strategies +**Through the UI** -#### Chunking Strategy Selection -```python -# Dynamic chunker selection based on config -chunker_mode = config.get("chunker_mode", "legacy") - -if chunker_mode == "docling": - self.chunker = DoclingChunker( - max_tokens=chunk_size, - overlap=overlap_sentences, - tokenizer_model="Qwen/Qwen3-Embedding-0.6B" - ) -else: - self.chunker = MarkdownRecursiveChunker( - max_chunk_size=chunk_size, - min_chunk_size=min(chunk_overlap, chunk_size // 4) - ) +1. `IndexForm` โ†’ `chatAPI.buildIndex()` โ†’ `POST :8000/indexes//build` (or `POST :8000/sessions//index` for session uploads). +2. `backend/server.py` normalises the option names (`INDEX_OPTIONS`, `server.py:58-70`), omits anything the caller did not send so pipeline defaults apply, and POSTs to `${RAG_API_URL}/index`. +3. `rag_system/api_server.py:392-473` validates the body, builds a per-request config from a deep copy of the pipeline profile (`_build_index_config_override`, `:197-236`), constructs a fresh `IndexingPipeline` with it, and calls `run(file_paths)`. +4. The chosen embedding model is written into the index's SQLite metadata (`api_server.py:445-449`); retrieval later re-applies it (`api_server.py:116-134`). + +**From the command line** + +```bash +python -m rag_system.main index /path/to/docs --mode default ``` -#### Recursive Markdown Chunking Algorithm -```python -def chunk(self, text: str, document_id: str, metadata: Dict) -> List[Dict]: - # Priority hierarchy for splitting - separators = [ - "\n\n# ", # H1 headers (highest priority) - "\n\n## ", # H2 headers - "\n\n### ", # H3 headers - "\n\n", # Paragraph breaks - "\n", # Line breaks - ". ", # Sentence boundaries - " " # Word boundaries (last resort) - ] - - chunks = [] - current_chunk = "" - - for separator in separators: - if len(current_chunk) <= self.max_chunk_size: - continue - - # Split on current separator - parts = current_chunk.split(separator) - - # Reassemble with overlap - for i, part in enumerate(parts): - if len(part) > self.max_chunk_size: - # Recursively split large parts - continue - - # Add overlap from previous chunk - if i > 0 and len(chunks) > 0: - overlap_text = chunks[-1]["text"][-self.chunk_overlap:] - part = overlap_text + separator + part - - chunks.append({ - "text": part, - "document_id": document_id, - "metadata": {**metadata, "chunk_index": len(chunks)} - }) +`_collect_file_paths` (`rag_system/main.py:133-146`) walks a directory for `.pdf`, `.docx`, `.html`, `.htm`, `.md`, `.txt`, or accepts a single file; the `index` branch (`main.py:171-188`) then calls `factory.get_indexing_pipeline(mode).run(file_paths)`. `--mode` accepts `default` or `fast`. `python rag_system/main.py โ€ฆ` as a file path does not work โ€” run it as a module from the project root. + +`create_index_script.py` is a separate interactive/batch tool that creates a SQLite index row and points the pipeline at that index's own table (`--batch `, `--config `, `--create-sample`). + +## Request fields (`POST /index`) + +Both camelCase and snake_case are accepted; the RAG API normalises to snake_case once at parse time (`api_server.py:51-76`). + +| Field | Default | Effect | +|-------|---------|--------| +| `file_paths` | **required** | List of absolute paths. A missing or non-list value returns HTTP 400. | +| `session_id` | โ€“ | Sets `overview_path` to `index_store/overviews/.jsonl`. | +| `table_name` | resolved from the session's index | Sets `storage.text_table_name` and `retrieval.dense.lancedb_table_name`. | +| `enable_latechunk` | `false` | Writes a second `
_lc` table of late-chunked vectors. | +| `enable_docling_chunk` | `true` | Maps to `chunker_mode: "docling"` when true, `"legacy"` when false. See the note below. | +| `chunk_size` | `512` | Token budget per chunk, for both chunkers. | +| `retrieval_mode` (alias `search_type`) | โ€“ | Validated against `hybrid` / `vector_only` / `fts_only` (HTTP 400 otherwise) and recorded on the index config as `retrieval.search_type`. It cannot change the artifacts written; it takes effect at query time. | +| `window_size` | `2` | Neighbour window for contextual enrichment. | +| `enable_enrich` | `true` | Contextual enrichment on/off. | +| `embedding_model` | profile value | Overrides `embedding_model_name` and is stamped into the index metadata. | +| `enrich_model` | โ€“ | Model for contextual enrichment. | +| `overview_model_name` (aliases `overviewModel`, `overview_model`) | โ€“ | Model for document overviews. | +| `batch_size_embed` | `50` | Embedding batch size. | +| `batch_size_enrich` | `25` | Enrichment batch size. | + +Response (`api_server.py:451-467`): + +```jsonc +{ + "message": "Indexing process for N file(s) completed successfully.", + "table_name": "text_pages_", + "latechunk": false, + "docling_chunk": true, + "indexing_config": { + "chunk_size": 512, "retrieval_mode": "hybrid", "window_size": 2, + "enable_enrich": true, "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": null, "overview_model_name": null, + "batch_size_embed": 50, "batch_size_enrich": 25 + } +} ``` -### DocLing Chunking Implementation +`indexing_config.embedding_model` reports the model actually used, not the raw request value. There is no `indexed_files` field. -#### Token-Aware Sentence Packing -```python -class DoclingChunker: - def __init__(self, max_tokens: int = 512, overlap: int = 1, - tokenizer_model: str = "Qwen/Qwen3-Embedding-0.6B"): - self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_model) - self.max_tokens = max_tokens - self.overlap = overlap # sentences of overlap - - def split_markdown(self, markdown: str, document_id: str, metadata: Dict): - sentences = self._sentence_split(markdown) - chunks = [] - window = [] - - while sentences: - # Add sentences until token limit - while (sentences and - self._token_len(" ".join(window + [sentences[0]])) <= self.max_tokens): - window.append(sentences.pop(0)) - - if not window: # Single sentence > limit - window.append(sentences.pop(0)) - - # Create chunk - chunk_text = " ".join(window) - chunks.append({ - "chunk_id": f"{document_id}_{len(chunks)}", - "text": chunk_text, - "metadata": { - **metadata, - "chunk_index": len(chunks), - "heading_path": metadata.get("heading_path", []), - "block_type": metadata.get("block_type", "paragraph") - } - }) - - # Add overlap for next chunk - if self.overlap and sentences: - overlap_sentences = window[-self.overlap:] - sentences = overlap_sentences + sentences - window = [] - - return chunks -``` +Unknown fields are ignored rather than rejected. -#### Document Structure Preservation -```python -def chunk_document(self, doc, document_id: str, metadata: Dict): - """Walk DoclingDocument tree and emit structured chunks.""" - chunks = [] - current_heading_path = [] - buffer = [] - - # Process document elements in reading order - for txt_item in doc.texts: - role = getattr(txt_item, "role", None) - - if role == "heading": - self._flush_buffer(buffer, chunks, current_heading_path) - level = getattr(txt_item, "level", 1) - # Update heading hierarchy - current_heading_path = current_heading_path[:level-1] - current_heading_path.append(txt_item.text.strip()) - continue - - # Accumulate text in token-aware buffer - text_piece = txt_item.text - if self._buffer_would_exceed_limit(buffer, text_piece): - self._flush_buffer(buffer, chunks, current_heading_path) - - buffer.append(text_piece) - - self._flush_buffer(buffer, chunks, current_heading_path) - return chunks -``` +## Pipeline config keys -### Contextual Enrichment Implementation +Read by `IndexingPipeline.__init__` and by `run()` for the table paths: -#### Batch Processing Pattern -```python -class ContextualEnricher: - def enrich_batch(self, chunks: List[Dict]) -> List[Dict]: - enriched_chunks = [] - - # Process in batches to manage memory - for i in range(0, len(chunks), self.batch_size): - batch = chunks[i:i + self.batch_size] - - # Parallel enrichment within batch - with concurrent.futures.ThreadPoolExecutor(max_workers=4) as executor: - futures = [ - executor.submit(self._enrich_single_chunk, chunk) - for chunk in batch - ] - - for future in concurrent.futures.as_completed(futures): - enriched_chunks.append(future.result()) - - return enriched_chunks -``` +| Key | Default in code | `default` profile | `fast` profile | +|-----|-----------------|-------------------|----------------| +| `chunker_mode` | `docling` | not set | not set | +| `chunking.chunk_size` (aliases `chunk_size`, `max_tokens`) | `1500` | not set | not set | +| `overlap_sentences` | `1` | not set | not set | +| `embedding_model_name` | `EXTERNAL_MODELS["embedding_model"]` | `microsoft/harrier-oss-v1-0.6b` | same | +| `storage.lancedb_uri` / `db_path` / `lancedb_path` | โ€“ (raises if all absent) | `./lancedb` | `./lancedb` | +| `storage.text_table_name` | falls back to `retrieval.dense.lancedb_table_name`, then `default_text_table` | `text_pages_v4` | `text_pages_v4` | +| `indexing.embedding_batch_size` | `50` | `50` | `100` | +| `indexing.enrichment_batch_size` | `10` | `10` | `50` | +| `retrieval.dense.enabled` | `true` | `true` | `true` | +| `retrieval.latechunk.enabled` (also read as `late_chunking`) | `false` | `true` | `false` | +| `retrieval.latechunk.lancedb_table_name` / `table_suffix` | suffix `_lc` | not set | not set | +| `contextual_enricher.enabled` / `window_size` | `false` / `1` | `true` / `1` | `false` / `1` | +| `enrich_model` (then `enrichment_model_name`, then `OLLAMA_CONFIG["enrichment_model"]`, then `generation_model`) | โ€“ | โ€“ | โ€“ | +| `overview.enabled` | `true` | not set โ‡’ on | not set โ‡’ on | +| `overview_model_name` / `overview.model` | falls back to enrichment then generation model | not set | not set | +| `overview_first_n_chunks` / `overview.max_chunks` | `5` | not set | not set | +| `overview_path` | `index_store/overviews/overviews.jsonl` | not set | not set | -#### Contextual Prompt Engineering -```python -def _generate_context_summary(self, chunk_text: str, surrounding_context: str) -> str: - prompt = f""" - Analyze this text chunk and provide a concise summary that captures: - 1. Main topics and key information - 2. Context within the broader document - 3. Relevance for search and retrieval - - Document Context: - {surrounding_context} - - Chunk to Analyze: - {chunk_text} - - Summary (max 2 sentences): - """ - - response = self.llm_client.complete( - prompt=prompt, - model=self.ollama_config["enrichment_model"] # qwen3:0.6b - ) - - return response.strip() -``` +Note the layering: the CLI/profile path uses the code default `chunk_size` of **1500** tokens because `PIPELINE_CONFIGS` sets no `chunk_size`, while the HTTP path always sends **512**. Likewise `enrichment_batch_size` is 10 from the profile but 25 over HTTP. -### Embedding Generation Pipeline +## Conversion (`ingestion/document_converter.py`) -#### Model Selection Strategy -```python -def select_embedder(model_name: str, ollama_host: str = None): - """Select appropriate embedder based on model name.""" - if "Qwen3-Embedding" in model_name: - return QwenEmbedder(model_name=model_name) - elif "bge-" in model_name: - return BGEEmbedder(model_name=model_name) - elif ollama_host and model_name in ["nomic-embed-text"]: - return OllamaEmbedder(model_name=model_name, host=ollama_host) - else: - # Default to Qwen embedder - return QwenEmbedder(model_name="Qwen/Qwen3-Embedding-0.6B") -``` +Three docling converters are built independently at construction time, so a failure in one does not disable the others (`:67-104`): -#### Batch Embedding Generation -```python -class QwenEmbedder: - def create_embeddings(self, texts: List[str]) -> np.ndarray: - """Generate embeddings in batches for efficiency.""" - embeddings = [] - - for i in range(0, len(texts), self.batch_size): - batch = texts[i:i + self.batch_size] - - # Tokenize and encode - inputs = self.tokenizer( - batch, - padding=True, - truncation=True, - max_length=512, - return_tensors='pt' - ) - - with torch.no_grad(): - outputs = self.model(**inputs) - # Mean pooling over token embeddings - batch_embeddings = outputs.last_hidden_state.mean(dim=1) - embeddings.append(batch_embeddings.cpu().numpy()) - - return np.vstack(embeddings) -``` +* **no-OCR** PDF converter, +* **OCR** PDF converter, +* **general** converter for DOCX/HTML/MD. -### LanceDB Storage Implementation +`.txt` files bypass docling entirely and are wrapped in a fenced code block (`:152-166`). -#### Table Management Strategy -```python -class LanceDBManager: - def create_table_if_not_exists(self, table_name: str, schema: Schema): - """Create LanceDB table with proper schema.""" - try: - table = self.db.open_table(table_name) - print(f"Table {table_name} already exists") - return table - except FileNotFoundError: - # Table doesn't exist, create it - table = self.db.create_table( - table_name, - schema=schema, - mode="create" - ) - print(f"Created new table: {table_name}") - return table - - def index_chunks(self, chunks: List[Dict], table_name: str): - """Store chunks with embeddings in LanceDB.""" - table = self.get_table(table_name) - - # Prepare data for insertion - records = [] - for chunk in chunks: - record = { - "chunk_id": chunk["chunk_id"], - "text": chunk["text"], - "vector": chunk["embedding"].tolist(), - "metadata": json.dumps(chunk["metadata"]), - "document_id": chunk["metadata"]["document_id"], - "chunk_index": chunk["metadata"]["chunk_index"] - } - records.append(record) - - # Batch insert - table.add(records) - - # Create vector index for fast similarity search - table.create_index("vector", config=IvfPq(num_partitions=256)) -``` +For PDFs, `_pdf_has_text()` opens the file with PyMuPDF and checks for any extractable text; if there is none, the OCR converter is used, otherwise the fast no-OCR converter (`:125-150`). When a PDF has no text layer but no OCR converter is available, the pipeline logs and retries without OCR rather than skipping the file. -### Overview Building for Query Routing +**OCR engine selection** (`build_ocr_options`, `:22-48`) picks the first engine whose backend is actually installed: -#### Document Summarization Strategy -```python -class OverviewBuilder: - def build_overview(self, chunks: List[Dict], document_id: str) -> Dict: - """Generate document overview for query routing.""" - # Take first N chunks for overview (usually most important) - sample_chunks = chunks[:self.max_chunks_for_overview] - combined_text = "\n\n".join([c["text"] for c in sample_chunks]) - - overview_prompt = f""" - Analyze this document and create a brief overview that includes: - 1. Main topic and purpose - 2. Key themes and concepts - 3. Document type and domain - 4. Relevant search keywords - - Document text: - {combined_text} - - Overview (max 3 sentences): - """ - - overview = self.llm_client.complete( - prompt=overview_prompt, - model=self.overview_model # qwen3:0.6b for speed - ) - - return { - "document_id": document_id, - "overview": overview.strip(), - "chunk_count": len(chunks), - "keywords": self._extract_keywords(combined_text), - "created_at": datetime.now().isoformat() - } - - def save_overview(self, overview: Dict): - """Save overview to JSONL file for query routing.""" - overview_path = f"./index_store/overviews/{overview['document_id']}.jsonl" - - with open(overview_path, 'w') as f: - json.dump(overview, f) -``` +| Order | docling options class | Requires | +|-------|----------------------|----------| +| 1 | `OcrMacOptions` | macOS only (`platform.system() == "Darwin"`) plus the `ocrmac` package | +| 2 | `EasyOcrOptions` | `easyocr` | +| 3 | `RapidOcrOptions` | `rapidocr` (or the older `rapidocr_onnxruntime` package name โ€” either is accepted) | +| 4 | `TesseractOcrOptions` | `tesserocr` | +| 5 | `TesseractCliOcrOptions` | a `tesseract` binary on `PATH` | -### Performance Optimizations +If none is available it logs "No OCR engine available; using docling's default OCR settings." On Linux/Docker, install one of the non-macOS backends if you need scanned-PDF support โ€” `ocrmac` is macOS-only and is excluded from `requirements-docker.txt`. -#### Memory Management -```python -class IndexingPipeline: - def __init__(self, config: Dict, ollama_client: OllamaClient, ollama_config: Dict): - # Lazy initialization to save memory - self._pdf_converter = None - self._chunker = None - self._embedder = None - - def _get_embedder(self): - """Lazy load embedder to avoid memory overhead.""" - if self._embedder is None: - model_name = self.config.get("embedding_model_name", "Qwen/Qwen3-Embedding-0.6B") - self._embedder = select_embedder(model_name) - return self._embedder - - def process_document_batch(self, file_paths: List[str]): - """Process documents in batches to manage memory.""" - for batch_start in range(0, len(file_paths), self.batch_size): - batch = file_paths[batch_start:batch_start + self.batch_size] - - # Process batch - self._process_batch(batch) - - # Cleanup to free memory - if hasattr(self, '_embedder') and self._embedder: - self._embedder.cleanup() -``` +Conversion returns `[(markdown, metadata, DoclingDocument)]` for docling paths and `[(markdown, metadata)]` for `.txt`. A conversion error is caught per document and returns `[]` (`:190-192`), so one bad file does not abort the run. -#### Parallel Processing -```python -def run_parallel_processing(self, file_paths: List[str]): - """Process multiple documents in parallel.""" - with concurrent.futures.ProcessPoolExecutor(max_workers=4) as executor: - futures = [] - - for file_path in file_paths: - future = executor.submit(self._process_single_file, file_path) - futures.append(future) - - # Collect results - results = [] - for future in concurrent.futures.as_completed(futures): - try: - result = future.result(timeout=300) # 5 minute timeout - results.append(result) - except Exception as e: - print(f"Error processing file: {e}") - - return results -``` +## Chunking -### Error Handling and Recovery +`chunker_mode` defaults to `"docling"` (`indexing_pipeline.py:27`). `MarkdownRecursiveChunker` is reached when `chunker_mode` is anything else, or when `DoclingChunker` construction raises (`:48-54`). -#### Graceful Degradation -```python -def run(self, file_paths: List[str], table_name: str): - """Main pipeline with comprehensive error handling.""" - processed_files = [] - failed_files = [] - - for file_path in file_paths: - try: - # Attempt processing - chunks = self._process_single_file(file_path) - - if chunks: - # Store successfully processed chunks - self._store_chunks(chunks, table_name) - processed_files.append(file_path) - else: - print(f"โš ๏ธ No chunks generated from {file_path}") - failed_files.append((file_path, "No chunks generated")) - - except Exception as e: - print(f"โŒ Error processing {file_path}: {e}") - failed_files.append((file_path, str(e))) - continue # Continue with other files - - # Return summary - return { - "processed": len(processed_files), - "failed": len(failed_files), - "processed_files": processed_files, - "failed_files": failed_files - } -``` +> **Note:** the RAG API maps `enable_docling_chunk` straight to `chunker_mode` โ€” `"docling"` when true, `"legacy"` when false (`api_server.py::_build_index_config_override`), and the HTTP default is `true`. The legacy chunker is also selectable by setting `chunker_mode: "legacy"` in a pipeline config (which `create_index_script.py` does). -#### Recovery Mechanisms -```python -def recover_from_partial_failure(self, table_name: str, document_id: str): - """Recover from partial indexing failures.""" - try: - # Check what was already processed - table = self.db_manager.get_table(table_name) - existing_chunks = table.search().where(f"document_id = '{document_id}'").to_list() - - if existing_chunks: - print(f"Found {len(existing_chunks)} existing chunks for {document_id}") - return True - - # Cleanup partial data - self._cleanup_partial_data(table_name, document_id) - return False - - except Exception as e: - print(f"Recovery failed: {e}") - return False -``` +**DoclingChunker** (`ingestion/docling_chunker.py`) -### Configuration and Customization +* `chunk_document(doc, โ€ฆ)` walks the `DoclingDocument` with docling's `iterate_items()`, which yields `(item, level)` in true reading order โ€” tables included inline. Items labelled `table` are exported to markdown and emitted as atomic chunks where they appear in the flow; items labelled `section_header` (with their `level`) maintain a `heading_path` and are not emitted as content; page numbers are taken from each item's `prov`; paragraph text is token-packed until `max_tokens`. Anything unexpected in the walk falls back to `split_markdown`. +* A second consolidation pass merges consecutive paragraph chunks that share a page and heading path, up to `max_tokens` (`:199-246`). +* `split_markdown()` is the fallback when only Markdown is available: it runs the legacy chunker with a 10,000-token ceiling, then repacks sentences to `max_tokens`, carrying `overlap` sentences (default 1) into the next window (`:47-83`). +* `chunk_document()` is what runs for documents converted by docling; the pipeline picks it via `hasattr(self.chunker, "chunk_document")` (`indexing_pipeline.py:192-195`). -#### Pipeline Configuration Options -```python -DEFAULT_CONFIG = { - "chunking": { - "strategy": "docling", # "docling", "recursive", "fixed" - "max_tokens": 512, - "overlap": 64, - "min_chunk_size": 100 - }, - "embedding": { - "model_name": "Qwen/Qwen3-Embedding-0.6B", - "batch_size": 32, - "max_length": 512 - }, - "enrichment": { - "enabled": True, - "model": "qwen3:0.6b", - "batch_size": 16 - }, - "overview": { - "enabled": True, - "max_chunks": 5, - "model": "qwen3:0.6b" - }, - "storage": { - "create_index": True, - "index_type": "IvfPq", - "num_partitions": 256 - } -} +**MarkdownRecursiveChunker** (`ingestion/chunking.py`) + +Recursively splits on `\n## `, `\n### `, `\n#### `, ```` ``` ````, `\n\n`, then on word boundaries, and merges adjacent pieces up to `max_chunk_size` while respecting `min_chunk_size` (`:34-124`). The pipeline passes `min_chunk_size = max(1, chunk_size // 4)`. + +Both chunkers count tokens with `AutoTokenizer.from_pretrained(embedding_model_name)`. If the tokenizer cannot be loaded they log a warning and fall back to a 4-characters-per-token approximation. + +There is **no chunk-overlap knob**. The docling path carries `overlap_sentences` (default 1) only in `split_markdown`; `chunk_document` emits non-overlapping consolidated blocks; the legacy path has no overlap logic at all. + +After chunking, the pipeline stamps a sequential `metadata.chunk_index` on every chunk of the document (`indexing_pipeline.py:202-205`) โ€” this is what context expansion and late-chunk merging use at query time. + +## Document overviews + +`OverviewBuilder.build_and_store(doc_id, chunks)` (`indexing/overview_builder.py:33-49`) runs once per document, inside the chunking loop and **before** enrichment. It sends the first `first_n_chunks` chunks (default 5), truncated to 5000 characters, to the overview model and appends one JSON line per document: + +```jsonc +{"doc_id": "report.pdf", "overview": "โ€ฆ"} ``` -#### Custom Processing Hooks -```python -class IndexingPipeline: - def __init__(self, config: Dict, hooks: Dict = None): - self.hooks = hooks or {} - - def _run_hook(self, hook_name: str, *args, **kwargs): - """Execute custom processing hooks.""" - if hook_name in self.hooks: - return self.hooks[hook_name](*args, **kwargs) - return None - - def process_chunk(self, chunk: Dict) -> Dict: - """Process single chunk with custom hooks.""" - # Pre-processing hook - chunk = self._run_hook("pre_chunk_process", chunk) or chunk - - # Standard processing - if self.contextual_enricher: - chunk = self.contextual_enricher.enrich_chunk(chunk) - - # Post-processing hook - chunk = self._run_hook("post_chunk_process", chunk) or chunk - - return chunk +Written in **append** mode to `overview_path`, which is one file per index (`index_store/overviews/.jsonl` when the API supplies a `session_id`), not one file per document. Failures are caught per document and logged (`indexing_pipeline.py:208-212`). Set `overview.enabled: false` in the pipeline config to skip the stage; there is no HTTP field for it. + +## Contextual enrichment + +`ContextualEnricher.enrich_chunks(chunks, window_size)` (`indexing/contextualizer.py:82-144`) prepends an LLM-written summary to each chunk's embedded text and preserves the untouched original in `metadata.original_text`: + ``` +Context: <2-5 sentence summary> --- -## Current Implementation Status - -### Completed Features โœ… -- DocLing-based PDF processing with OCR fallback -- Multiple chunking strategies (DocLing, Recursive, Fixed-size) -- Qwen3-Embedding-0.6B integration -- Contextual enrichment with qwen3:0.6b -- LanceDB storage with vector indexing -- Overview generation for query routing -- Batch processing and parallel execution -- Comprehensive error handling - -### In Development ๐Ÿšง -- Graph extraction and knowledge graph building -- Multimodal processing for images and tables -- Advanced late-chunking optimization -- Distributed processing support - -### Planned Features ๐Ÿ“‹ -- Custom model fine-tuning pipeline -- Real-time incremental indexing -- Cross-document relationship extraction -- Advanced metadata enrichment + +``` ---- +Processing is **sequential**: `BatchProcessor.process_in_batches` is a plain `for` loop over slices with progress reporting and a `gc.collect()` every fifth batch (`utils/batch_processor.py:105-125`). "Batch size" controls reporting granularity and memory, not concurrency. Budget one LLM round-trip per chunk when sizing a run โ€” there is no thread or process pool anywhere in the indexing path. -## Performance Benchmarks +Chain-of-thought markers (`โ€ฆ`), assistant tags and a leading `Answer:` are stripped from the summary; a summary shorter than 5 characters is discarded and the chunk is indexed unenriched (`contextualizer.py:56-74`). -| Document Type | Processing Speed | Memory Usage | Storage Efficiency | -|---------------|------------------|--------------|-------------------| -| Text PDFs | 2-5 pages/sec | 2-4GB | 1MB/100 pages | -| Image PDFs | 0.5-1 page/sec | 4-8GB | 2MB/100 pages | -| Technical Docs | 1-3 pages/sec | 3-6GB | 1.5MB/100 pages | -| Research Papers | 2-4 pages/sec | 2-4GB | 1.2MB/100 pages | +## Embedding -## Extension Points +`select_embedder(model_name, ollama_host)` (`indexing/representations.py:167-173`) is a two-way dispatch: -### Custom Chunkers -```python -class CustomChunker(BaseChunker): - def chunk(self, text: str, document_id: str, metadata: Dict) -> List[Dict]: - # Implement custom chunking logic - pass -``` +* the name contains `/` or starts with `http` โ†’ `QwenEmbedder` (HuggingFace `AutoModel`, loaded in-process); +* anything else โ†’ `OllamaEmbedder`, one HTTP call to `/api/embeddings` per text. -### Custom Embedders -```python -class CustomEmbedder(BaseEmbedder): - def create_embeddings(self, texts: List[str]) -> np.ndarray: - # Implement custom embedding generation - pass -``` +`QwenEmbedder` (`representations.py:15-94`): -### Custom Enrichers -```python -class CustomEnricher(BaseEnricher): - def enrich_chunk(self, chunk: Dict) -> Dict: - # Implement custom enrichment logic - pass -``` \ No newline at end of file +* device order CUDA โ†’ MPS โ†’ CPU; `float16` off CPU; +* weights cached per model name in a module-level dict, so repeated construction reuses them; +* tokenizer is loaded with `padding_side="left"`, and pooling takes the **last real token**, not a mean โ€” with left padding that is simply the final column, and the right-padding case is handled explicitly (`:66-76`); +* `max_length` is `min(tokenizer.model_max_length, 8192)` so `truncation=True` actually truncates; +* NaN/Inf values are replaced with zeros and logged. + +`LateChunkEncoder` mean-pools over each span instead (`indexing/latechunk.py:82`) โ€” the two paths deliberately use different pooling and write to different tables. + +**Embedding dimensions are never hard-coded.** `VectorIndexer.index()` reads the width from the first produced vector (`indexing/embedders.py:47`) and builds the pyarrow schema from it. If the target table already exists with a different width, indexing raises: + +> Table 'โ€ฆ' stores N-dim vectors but the current embedding model produced M-dim vectors. Changing the embedding model requires rebuilding the index. + +**Changing `EMBEDDING_MODEL` (or the per-request `embedding_model`) requires re-indexing.** `Qwen/Qwen3-Embedding-4B` produces 2560-dim vectors; `microsoft/harrier-oss-v1-0.6b` and `Qwen/Qwen3-Embedding-0.6B` both produce 1024-dim. They are not interchangeable against an existing table โ€” and because a matching width does not make two models compatible, `VectorIndexer` also stamps the embedding model name (and an L2-`normalized` flag) into each table's metadata and refuses to write or read it with a different one. + +## LanceDB write and full-text index + +`VectorIndexer.index(table_name, chunks, embeddings)` (`indexing/embedders.py:39-135`) writes one row per chunk: + +| Column | Content | +|--------|---------| +| `vector` | fixed-size float32 list, width from the model | +| `text` | the text that was embedded (enriched, if enrichment ran) | +| `chunk_id`, `document_id`, `chunk_index` | flattened identifiers used by context expansion | +| `metadata` | the whole chunk dict as a JSON string, including `original_text` | + +Chunks whose vector contains NaN or Inf are skipped with a warning. The table is created when it does not exist; when it does, re-indexing a document first deletes that document's existing rows (delete-by-`document_id` before append โ€” `VectorIndexer.index`), so re-running a build **replaces** the previous chunks instead of duplicating them. The `
_lc` late-chunk table gets the same treatment. `tbl.add(..., on_bad_vectors='drop')` is retried with a zero-fill strategy on failure. + +Immediately after the vector write, the pipeline ensures a Lance native full-text index on the `text` column (`indexing_pipeline.py:264-284`), guarding against both the LanceDB default name `text_idx` and this project's older name `fts_text` so a rebuild does not raise. This index is what the `hybrid` and `fts_only` retrieval modes query โ€” there is no separate BM25 store on disk and no `bm25_path` config key. + +No ANN index is created. `create_index` / IVF-PQ appears nowhere in `rag_system/`, so vector search is an exhaustive scan. That is fine for the corpus sizes this project targets and avoids the training-set-size requirements of IVF-PQ. + +## Late chunking (optional) + +When `retrieval.latechunk.enabled` is true, a second pass runs per document (`indexing_pipeline.py:286-327`): + +1. Concatenate the document's chunk texts with newlines and record each chunk's character span. +2. Feed the whole document through `LateChunkEncoder.encode()` โ€” one forward pass, truncated at 8192 tokens โ€” and mean-pool the token hidden states inside each span. +3. Write those vectors to `
_lc` (or `latechunk.lancedb_table_name` when set) with the same chunk rows. + +Each chunk vector is therefore produced with knowledge of the whole document. A per-document encode failure, or a vector/chunk count mismatch, logs a warning and skips that document. The cost is a second full copy of the embedding model in memory and roughly double the vectors written. + +The retrieval side reads `
_lc` with the same default suffix โ€” see `retrieval_pipeline.md`. + +## Knowledge graph โ€” removed 2026-08-09 + +There is no knowledge-graph step any more. `indexing/graph_extractor.py`, the +`retrieval.graph.*` config keys and the `.gml` writer were deleted at roadmap +item 2.5. The path had never been armed (both shipped profiles set +`enabled: false`), and the evidence argues against reviving it: GraphRAG *loses* +on single-hop retrieval, its multi-hop gains span +3 to +27 points depending on +how well the vector baseline is tuned, and it costs **41โ€“57ร— at indexing** and up +to **~377ร— in query tokens** โ€” see +[`research/academic-evidence-2026.md`](research/academic-evidence-2026.md) ยง6. +`networkx` was dropped from `requirements.txt` in the same change. + +An index built before this change is unaffected: the graph lived in a standalone +`.gml` file that nothing else read. A stale `index_store/graph/` directory can be +deleted by hand. + +## Error handling + +* **Per file** โ€” conversion or chunking errors are caught, logged as `โŒ Error processing `, counted by the progress tracker, and the run continues (`indexing_pipeline.py:219-222`). +* **No chunks at all** โ€” the run raises `RuntimeError` ("No text chunks were generated from the supplied documentsโ€ฆ"), which surfaces as a 500 from `POST /index` so a failed conversion is never reported as a successful build (`:226-231`). +* **Duplicate table** โ€” `VectorIndexer` appends instead of recreating (`embedders.py:105-120`); the backend additionally treats an "already exists" error from the RAG API as non-fatal and reports "Index already built โ€“ skipping rebuild." (`backend/server.py`, `handle_build_index`). +* **Dimension mismatch** โ€” hard `ValueError`, on purpose. Silently dropping or recreating an index would corrupt it. +* **FTS index** โ€” creation failures are logged, not raised; the vectors are already written and `vector_only` retrieval still works. +* **Overview / late-chunk / enrichment failures** โ€” logged and skipped; the main vector index is still produced. + +At the end, `_print_final_statistics` reports files processed, chunks generated, average chunks per file, which components ran, and the batch sizes used (`indexing_pipeline.py:349-372`). + +## Not integrated + +* **Vision / multimodal.** There is no vision model anywhere in the pipeline: no image embeddings are produced and no image table is written. PDF understanding is docling's layout parsing plus OCR. Models such as GLM-OCR or Qwen3-VL would be reasonable extensions, but no code path exists for them today. +* **Parallel document processing.** Documents are processed one at a time and enrichment is one sequential LLM call per chunk. There is no `ProcessPoolExecutor` or `ThreadPoolExecutor` in the indexing path. + +--- +_Keep this document updated when stages, config keys, or the `/index` contract change._ diff --git a/Documentation/installation_guide.md b/Documentation/installation_guide.md index 2bb881ab..58c7e9b8 100644 --- a/Documentation/installation_guide.md +++ b/Documentation/installation_guide.md @@ -1,22 +1,23 @@ -# ๐Ÿ“ฆ RAG System Installation Guide +# ๐Ÿ“ฆ LocalGPT Installation Guide -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ -This guide provides step-by-step instructions for installing and setting up the RAG system using either Docker or direct development approaches. +This guide provides step-by-step instructions for installing and setting up +LocalGPT using either Docker or a direct development install. --- ## ๐ŸŽฏ Installation Options -### Option 1: Docker Deployment (Production Ready) ๐Ÿณ -- **Best for**: Production environments, isolated setups, easy management -- **Requirements**: Docker Desktop + Local Ollama -- **Setup time**: ~10 minutes +### Option 1: Docker Deployment ๐Ÿณ +- **Best for**: reproducible environments, keeping Python dependencies isolated +- **Requirements**: Docker Desktop + Ollama (host or containerized) +- **Setup time**: ~10 minutes plus image build and model downloads -### Option 2: Direct Development (Developer Friendly) ๐Ÿ’ป -- **Best for**: Development, customization, debugging +### Option 2: Direct Development ๐Ÿ’ป +- **Best for**: development, customization, debugging - **Requirements**: Python + Node.js + Ollama -- **Setup time**: ~15 minutes +- **Setup time**: ~15 minutes plus model downloads --- @@ -28,26 +29,32 @@ This guide provides step-by-step instructions for installing and setting up the - **CPU**: 4 cores, 2.5GHz+ - **RAM**: 8GB (16GB recommended) - **Storage**: 50GB free space -- **OS**: macOS 10.15+, Ubuntu 20.04+, Windows 10+ +- **OS**: macOS 10.15+, Ubuntu 20.04+, Windows 10+ (WSL2) #### **Recommended Requirements** - **CPU**: 8+ cores, 3.0GHz+ -- **RAM**: 32GB+ (for large models) +- **RAM**: 32GB+ (for the larger generation models) - **Storage**: 200GB+ SSD -- **GPU**: NVIDIA GPU with 8GB+ VRAM (optional) +- **GPU**: NVIDIA GPU with 8GB+ VRAM (optional; Apple Silicon uses MPS) + +Disk budget for the defaults: `qwen3.5:9b` and `qwen3.5:4b` in Ollama, plus +`microsoft/harrier-oss-v1-0.6b` (~1.2GB) in the HuggingFace cache +(`~/.cache/huggingface`). Reranking is on by default, but the reranker loads +lazily โ€” `Qwen/Qwen3-Reranker-4B` (~7.5GB) is fetched on the first reranked +query, not at startup. ### 1.2 Common Dependencies **Required for both approaches:** -- **Ollama**: AI model runtime (always required) -- **Git**: 2.30+ for cloning repository +- **Ollama**: model runtime (always required) +- **Git**: 2.30+ for cloning the repository **Docker-specific:** -- **Docker Desktop**: 24.0+ with Docker Compose +- **Docker Desktop**: 24.0+ with the Compose plugin **Direct Development-specific:** -- **Python**: 3.8+ -- **Node.js**: 16+ with npm +- **Python**: 3.10+ (3.11 recommended โ€” the Docker images use `python:3.11-slim`) +- **Node.js**: 20+ with npm --- @@ -70,23 +77,26 @@ ollama --version # Run the installer and follow setup wizard ``` -### 2.2 Configure Ollama +### 2.2 Pull the Models ```bash # Start Ollama server ollama serve -# In another terminal, install required models -ollama pull qwen3:0.6b # Fast model (650MB) -ollama pull qwen3:8b # High-quality model (4.7GB) +# In another terminal, install the two default models +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, decomposition, enrichment, verification # Verify models are installed ollama list # Test Ollama -ollama run qwen3:0.6b "Hello, how are you?" +ollama run qwen3.5:4b "Hello, how are you?" ``` +The embedding and reranker models are **not** Ollama models โ€” they are downloaded +from HuggingFace automatically the first time the pipeline loads them. + **โš ๏ธ Important**: Keep Ollama running (`ollama serve`) for the entire setup process. --- @@ -135,33 +145,41 @@ docker compose version 3. Restart computer and start Docker Desktop 4. Verify in PowerShell: `docker --version` -### 3.2 Clone and Setup RAG System +### 3.2 Clone and Start ```bash # Clone repository -git clone -cd rag_system_old +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT -# Verify Ollama is running +# Verify Ollama is running on the host curl http://localhost:11434/api/tags -# Start Docker containers +# Start Docker containers (uses local Ollama) ./start-docker.sh -# Wait for containers to start (2-3 minutes) -sleep 120 +# Or, without host Ollama: +# ./start-docker.sh container +# For scripts and CI, add -y so the script never prompts: +# ./start-docker.sh local -y (or: NONINTERACTIVE=1 ./start-docker.sh) # Verify deployment ./start-docker.sh status ``` +The first build compiles the frontend and installs the Python dependencies, and +`rag-api` loads the embedding model before it reports healthy (the reranker loads +lazily, on the first reranked query). The +`backend` service has `depends_on: rag-api: service_healthy`, so it deliberately +waits. Expect several minutes on a cold start. + ### 3.3 Test Docker Deployment ```bash # Test all endpoints curl -f http://localhost:3000 && echo "โœ… Frontend OK" curl -f http://localhost:8000/health && echo "โœ… Backend OK" -curl -f http://localhost:8001/models && echo "โœ… RAG API OK" +curl -f http://localhost:8001/health && echo "โœ… RAG API OK" curl -f http://localhost:11434/api/tags && echo "โœ… Ollama OK" # Access the application @@ -177,8 +195,8 @@ open http://localhost:3000 #### **Python Setup:** ```bash # Clone repository -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Create virtual environment (recommended) python -m venv venv @@ -194,17 +212,27 @@ pip install -r requirements.txt python -c "import torch; print('โœ… PyTorch OK')" python -c "import transformers; print('โœ… Transformers OK')" python -c "import lancedb; print('โœ… LanceDB OK')" +python -c "import docling; print('โœ… Docling OK')" ``` +`requirements.txt` at the repository root is the full install used by +`run_system.py`. Two variants exist for narrower cases: + +| File | Purpose | +|------|---------| +| `requirements.txt` | Everything the local stack needs | +| `requirements-docker.txt` | Used by `Dockerfile.backend` and `Dockerfile.rag-api` (no macOS-only packages) | +| `rag_system/requirements.txt` | RAG core only; adds macOS `ocrmac` and `ibm-watsonx-ai` | +| `backend/requirements.txt` | Gateway only (`requests`, `python-dotenv`) | + #### **Node.js Setup:** ```bash # Install Node.js dependencies npm install # Verify Node.js setup -node --version # Should be 16+ +node --version # Should be 20+ npm --version -npm list --depth=0 ``` ### 4.2 Start Direct Development @@ -216,25 +244,30 @@ curl http://localhost:11434/api/tags # Start all components with one command python run_system.py -# Or start components manually in separate terminals: +# Or start components manually, each from the repository root: # Terminal 1: python -m rag_system.api_server -# Terminal 2: cd backend && python server.py +# Terminal 2: python backend/server.py # Terminal 3: npm run dev ``` +> Always run from the repository root. `backend/chat_data.db`, `lancedb/`, +> `index_store/` and `shared_uploads/` are resolved relative to the working +> directory, so `cd backend && python server.py` would create a second database at +> `backend/backend/chat_data.db`. + ### 4.3 Test Direct Development ```bash -# Check system health +# Deep check: imports, config, LanceDB, embedding model, sample query python system_health_check.py +# HTTP health check per service +python run_system.py --health + # Test endpoints curl -f http://localhost:3000 && echo "โœ… Frontend OK" curl -f http://localhost:8000/health && echo "โœ… Backend OK" -curl -f http://localhost:8001/models && echo "โœ… RAG API OK" - -# Access the application -open http://localhost:3000 +curl -f http://localhost:8001/health && echo "โœ… RAG API OK" ``` --- @@ -245,54 +278,106 @@ open http://localhost:3000 ```bash # Clone repository -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Check repository structure ls -la -# Create required directories +# These are created on demand, but you can pre-create them mkdir -p lancedb index_store shared_uploads logs backend -touch backend/chat_data.db # Set permissions chmod -R 755 lancedb index_store shared_uploads -chmod 664 backend/chat_data.db ``` +The SQLite file is created automatically the first time `ChatDatabase` is +constructed โ€” you do not need to `touch` it. + ### 5.2 Configuration +LocalGPT runs with no configuration file. Every variable below has a default that +matches the code. Override them in the shell, in a `.env` at the repository root +(`load_dotenv()` runs on import of `rag_system/main.py` and +`rag_system/factory.py`), or in `docker.env` for containers. `.env.example` ships +the same list. + #### **Environment Variables** -For Docker (automatic via `docker.env`): + +| Variable | Default | Read by | +|----------|---------|---------| +| `OLLAMA_HOST` | `http://localhost:11434` | `rag_system/main.py`, `backend/ollama_client.py` | +| `RAG_API_URL` | `http://localhost:8001` | `backend/server.py` | +| `NEXT_PUBLIC_API_URL` | `http://localhost:8000` | `src/lib/api.ts` (build-time) | +| `NEXT_PUBLIC_RAG_API_URL` | `http://localhost:8001` | `src/lib/api.ts` (build-time) | +| `DB_PATH` | `backend/chat_data.db` | `backend/database.py` | +| `LANCEDB_PATH` | `./lancedb` | `rag_system/main.py` (pipeline profiles), `backend/database.py`, `system_health_check.py` | +| `GENERATION_MODEL` | `qwen3.5:9b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | same | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | `rag_system/main.py` | +| `RERANKER_MODEL` | `Qwen/Qwen3-Reranker-4B` (loaded lazily on the first reranked query) | `rag_system/main.py` | +| `RAG_CONFIG_MODE` | `default` | `rag_system/api_server.py` (`default` or `fast`) | +| `RAG_API_TIMEOUT` | `600` | `backend/server.py` | +| `RAG_API_INDEX_TIMEOUT` | `3600` | `backend/server.py` | +| `LLM_BACKEND` | `ollama` | `rag_system/main.py` (`ollama` or `watsonx`) | +| `HF_TOKEN` | unset | HuggingFace downloads | + +For Docker these are set in `docker.env` and passed with +`docker compose --env-file docker.env`: + ```bash OLLAMA_HOST=http://host.docker.internal:11434 NODE_ENV=production RAG_API_URL=http://rag-api:8001 NEXT_PUBLIC_API_URL=http://localhost:8000 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 +DB_PATH=/app/backend/chat_data.db +LANCEDB_PATH=/app/lancedb +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B ``` -For Direct Development (set automatically by `run_system.py`): -```bash -OLLAMA_HOST=http://localhost:11434 -RAG_API_URL=http://localhost:8001 -NEXT_PUBLIC_API_URL=http://localhost:8000 -``` +`run_system.py` does **not** invent environment variables. It inherits your shell +environment unchanged and only adds `NODE_ENV=production` to the Python services +in `--mode prod`. If you want a non-default `OLLAMA_HOST` or `RAG_API_URL`, export +it before launching. + +`NEXT_PUBLIC_*` values are inlined into the frontend bundle by `next build`, so +they are build-time settings. In Docker they are passed as build args in +`docker-compose.yml`; changing them means `docker compose build frontend`. #### **Model Configuration** -The system defaults to these models: -- **Embedding**: `Qwen/Qwen3-Embedding-0.6B` (1024 dimensions) -- **Generation**: `qwen3:0.6b` for fast responses, `qwen3:8b` for quality -- **Reranking**: Built-in cross-encoder + +| Role | Default | Documented options | +|------|---------|--------------------| +| Generation | `qwen3.5:9b` | `qwen3.6:27b` (high-end, ~17GB), `qwen3.5:4b` (light) | +| Enrichment / utility | `qwen3.5:4b` | `qwen3.5:2b` (light) | +| Embedding | `microsoft/harrier-oss-v1-0.6b` (MIT, 1024 dims) | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context), `Qwen/Qwen3-Embedding-0.6B` (1024 dims) | +| Reranker (**on by default**) | `Qwen/Qwen3-Reranker-4B`, loaded lazily by the in-repo `QwenRerankerScorer` | `BAAI/bge-reranker-v2-m3`, `answerdotai/answerai-colbert-small-v1`, `Qwen/Qwen3-Reranker-0.6B` | + +Notes: +- Embedding dimensions are read from the loaded model, never hardcoded. + **Changing `EMBEDDING_MODEL` requires rebuilding every existing index** โ€” + appending vectors of a different width to a LanceDB table raises an explicit + error telling you to re-index. +- If the reranker fails to load, the pipeline logs a warning and continues + **without reranking**. There is no secondary reranker to fall back to. +- Any name containing `/` is treated as a HuggingFace model; anything else is + treated as an Ollama tag, so an Ollama embedding model such as + `nomic-embed-text` also works if you have pulled it. +- Vision / multimodal models are not wired into any pipeline. PDF parsing and OCR + are handled by Docling. ### 5.3 Database Initialization ```bash -# Initialize SQLite database +# Initialize SQLite database (also happens automatically at first use) python -c " from backend.database import ChatDatabase db = ChatDatabase() -db.init_database() -print('โœ… Database initialized') +print('โœ… Database initialized at', db.db_path) " # Verify database @@ -313,26 +398,27 @@ docker compose ps # For Direct development python system_health_check.py +python run_system.py --health # Universal health check curl -f http://localhost:3000 && echo "โœ… Frontend OK" curl -f http://localhost:8000/health && echo "โœ… Backend OK" -curl -f http://localhost:8001/models && echo "โœ… RAG API OK" +curl -f http://localhost:8001/health && echo "โœ… RAG API OK" curl -f http://localhost:11434/api/tags && echo "โœ… Ollama OK" ``` #### **RAG System Test:** ```bash -# Test RAG system initialization +# Test RAG system initialization (factory.py is the single entry point) python -c " -from rag_system.main import get_agent +from rag_system.factory import get_agent agent = get_agent('default') print('โœ… RAG System initialized successfully') " -# Test embedding generation +# Test embedding generation and report the real dimension python -c " -from rag_system.main import get_agent +from rag_system.factory import get_agent agent = get_agent('default') embedder = agent.retrieval_pipeline._get_text_embedder() test_emb = embedder.create_embeddings(['Hello world']) @@ -356,7 +442,8 @@ curl -X POST http://localhost:8000/sessions \ -H "Content-Type: application/json" \ -d '{"title": "Test Session"}' -# Test models endpoint +# Test models endpoints +curl http://localhost:8000/models curl http://localhost:8001/models # Test health endpoints @@ -375,13 +462,13 @@ curl http://localhost:8001/health # Ollama not responding curl http://localhost:11434/api/tags -# If fails, restart Ollama +# If it fails, restart Ollama pkill ollama ollama serve # Reinstall models if needed -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` #### **Docker Issues:** @@ -400,7 +487,7 @@ docker system prune -f #### **Python Issues:** ```bash # Check Python version -python --version # Should be 3.8+ +python --version # 3.10+ required, 3.11 recommended # Check virtual environment which python @@ -413,13 +500,24 @@ pip install -r requirements.txt --force-reinstall #### **Node.js Issues:** ```bash # Check Node version -node --version # Should be 16+ +node --version # Should be 20+ # Clear and reinstall rm -rf node_modules package-lock.json npm install ``` +#### **Scanned PDFs produce no text:** +A PDF with no text layer is re-run through Docling's OCR pipeline, but only with an +engine that is actually installed. `rag_system/ingestion/document_converter.py` +probes, in order: `ocrmac` (macOS only), `easyocr`, `rapidocr` (or the older +`rapidocr_onnxruntime` package name โ€” either is accepted), +`tesserocr`, then the `tesseract` binary. Install whichever suits your platform โ€” +`pip install ocrmac` on macOS (it is listed in `rag_system/requirements.txt` but not +in the root `requirements.txt`), or `pip install easyocr` / `apt install tesseract-ocr` +elsewhere. At startup the converter prints the engine it chose (`OCR engine: โ€ฆ`) or +`No OCR engine available; using docling's default OCR settings.` + ### 7.2 Performance Issues #### **Memory Problems:** @@ -431,35 +529,45 @@ vm_stat # macOS # For Docker: Increase memory allocation # Docker Desktop โ†’ Settings โ†’ Resources โ†’ Memory โ†’ 8GB+ -# Use smaller models -ollama pull qwen3:0.6b # Instead of qwen3:8b +# Use lighter models +export GENERATION_MODEL=qwen3.5:4b +# (the default embedder is already the small one, ~1.2GB) ``` #### **Slow Performance:** -- Use SSD storage for databases (`lancedb/`, `shared_uploads/`) -- Increase CPU cores if possible -- Close unnecessary applications -- Use smaller batch sizes in configuration +- Use SSD storage for `lancedb/` and `shared_uploads/` +- Switch to `RAG_CONFIG_MODE=fast` (vector-only retrieval, no reranking, + no decomposition, no verification) +- Lower `indexing.embedding_batch_size` / `enrichment_batch_size` if you are + swapping rather than compute-bound +- Remember the RAG API serialises requests โ€” one query or indexing run at a time --- ## 8. Post-Installation Setup -### 8.1 Model Optimization +### 8.1 Model Experiments ```bash -# Install additional models (optional) -ollama pull nomic-embed-text # Alternative embedding model -ollama pull llama3.1:8b # Alternative generation model +# Install additional generation models +ollama pull qwen3.6:27b # highest quality, ~17GB +ollama pull qwen3.5:2b # lightest utility model -# Test model switching +# Try one for a single request curl -X POST http://localhost:8001/chat \ -H "Content-Type: application/json" \ - -d '{"query": "Hello", "model": "qwen3:8b"}' + -d '{"query": "Hello", "model": "qwen3.5:4b"}' ``` +A per-request `model` is applied only for that request and then restored, and is +rejected with a warning if it is not valid for the active `LLM_BACKEND`. + ### 8.2 Security Configuration +Neither server implements authentication, and both send +`Access-Control-Allow-Origin: *`. Do not expose ports 8000/8001 outside a trusted +network. + ```bash # Set proper file permissions chmod 600 backend/chat_data.db # Restrict database access @@ -492,32 +600,35 @@ EOF chmod +x backup_system.sh ``` +All four paths are plain host directories (bind-mounted into the containers), so +a file copy is a complete backup. Stop the services first so SQLite and LanceDB +are not mid-write. + --- ## 9. Success Criteria ### 9.1 Installation Complete When: -- โœ… All health checks pass without errors +- โœ… `python run_system.py --health` exits 0 - โœ… Frontend loads at http://localhost:3000 -- โœ… All models are installed and responding -- โœ… You can create document indexes +- โœ… `ollama list` shows `qwen3.5:9b` and `qwen3.5:4b` +- โœ… You can create a document index - โœ… You can chat with uploaded documents -- โœ… No error messages in logs/terminal +- โœ… No error messages in `logs/` or the terminal -### 9.2 Performance Benchmarks +### 9.2 Performance Expectations -**Acceptable Performance:** -- System startup: < 5 minutes -- Index creation: < 2 minutes per 100MB document -- Query response: < 30 seconds -- Memory usage: < 8GB total +Numbers depend heavily on hardware, model size and document length. On a machine +matching the recommended requirements, expect: -**Optimal Performance:** -- System startup: < 2 minutes -- Index creation: < 1 minute per 100MB document -- Query response: < 10 seconds -- Memory usage: < 4GB total +- First startup dominated by model downloads (Ollama tags + the ~1.2GB embedding + model from HuggingFace); subsequent startups load from cache. The reranker + (~7.5GB) is downloaded separately, lazily on the first reranked query +- Indexing dominated by contextual enrichment โ€” it runs one LLM call per chunk, so + turn it off (`enable_enrich: false`) for the fastest ingest +- Query latency dominated by generation; `RAG_CONFIG_MODE=fast` removes reranking, + decomposition and verification --- @@ -525,18 +636,20 @@ chmod +x backup_system.sh ### 10.1 Getting Started -1. **Upload Documents**: Create your first index with PDF documents -2. **Explore Features**: Try different query types and models -3. **Customize**: Adjust model settings and chunk sizes +1. **Upload Documents**: Create your first index +2. **Explore Features**: Try different retrieval modes and models +3. **Customize**: Adjust chunk size, enrichment and verification per request 4. **Scale**: Add more documents and create multiple indexes ### 10.2 Additional Resources - **Quick Start**: See `Documentation/quick_start.md` - **Docker Usage**: See `Documentation/docker_usage.md` +- **Deployment**: See `Documentation/deployment_guide.md` - **System Architecture**: See `Documentation/architecture_overview.md` - **API Reference**: See `Documentation/api_reference.md` +- **WatsonX backend**: See `WATSONX_README.md` --- -**Congratulations! ๐ŸŽ‰** Your RAG system is now ready to use. Visit http://localhost:3000 to start chatting with your documents. \ No newline at end of file +**Congratulations! ๐ŸŽ‰** Visit http://localhost:3000 to start chatting with your documents. diff --git a/Documentation/prompt_inventory.md b/Documentation/prompt_inventory.md index a1f14a0d..cc6db547 100644 --- a/Documentation/prompt_inventory.md +++ b/Documentation/prompt_inventory.md @@ -1,70 +1,76 @@ # ๐Ÿ“œ Prompt Inventory (Ground-Truth) -_All generation / verification prompts currently hard-coded in the codebase._ -_Last updated: 2025-07-06_ +_Every prompt hard-coded in the codebase, re-derived from the current source._ -> Edit process: if you change a prompt in code, please **update this file** or, once we migrate to the central registry, delete the entry here. +> Edit process: if you change a prompt in code, update the line range here in the same commit. +> Line numbers drift as the code moves; the function and variable names are the stable references. ---- +## Which model runs which prompt -## 1. Indexing / Context Enrichment +There are two model roles (`rag_system/main.py` `OLLAMA_CONFIG`): -| ID | File & Lines | Variable / Builder | Purpose | -|----|--------------|--------------------|---------| -| `overview_builder.default` | `rag_system/indexing/overview_builder.py` `12-21` | `DEFAULT_PROMPT` | Generate 1-paragraph document overview for search-time routing. -| `contextualizer.system` | `rag_system/indexing/contextualizer.py` `11` | `SYSTEM_PROMPT` | System instruction: explain summarisation role. -| `contextualizer.local_context` | same file `13-15` | `LOCAL_CONTEXT_PROMPT_TEMPLATE` | Human message โ€“ wraps neighbouring chunks. -| `contextualizer.chunk` | same file `17-19` | `CHUNK_PROMPT_TEMPLATE` | Human message โ€“ shows the target chunk. -| `graph_extractor.entities` | `rag_system/indexing/graph_extractor.py` `20-31` | `entity_prompt` | Ask LLM to list entities. -| `graph_extractor.relationships` | same file `53-64` | `relationship_prompt` | Ask LLM to list relationships. +| Role | Config key | Default | Env override | +|------|-----------|---------|--------------| +| Generation โ€” user-facing answers | `generation_model` | `qwen3.5:9b` | `GENERATION_MODEL` | +| Utility โ€” routing, triage, decomposition, enrichment, overviews, verification | `enrichment_model` | `qwen3.5:4b` | `ENRICHMENT_MODEL` | -## 2. Retrieval / Query Transformation +Inside the agent the utility model is resolved by `Agent._utility_model()` (`rag_system/agent/loop.py:56-58`), which returns `ollama_config["enrichment_model"]` and falls back to `generation_model` when that key is absent. No prompt hard-codes a model name. -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `query_transformer.expand` | `rag_system/retrieval/query_transformer.py` `10-26` | Produce query rewrites (keywords, boolean). | -| `hyde.hypothetical_doc` | same `115-122` | HyDE hypothetical document generator. | -| `graph_query.translate` | same `124-140` | Translate user question to JSON KG query. | +--- -## 3. Pipeline Answer Synthesis +## 1. Indexing / context enrichment -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `retrieval_pipeline.synth_final` | `rag_system/pipelines/retrieval_pipeline.py` `217-256` | Turn verified facts into answer (with directives 1-6). | +| ID | File & lines | Variable / builder | Model | Purpose | +|----|--------------|--------------------|-------|---------| +| `overview_builder.default` | `rag_system/indexing/overview_builder.py` `13-19` | `OverviewBuilder.DEFAULT_PROMPT` | overview model (resolved at `indexing_pipeline.py:128-133`; utility model by default) | One-paragraph document overview used by the triage routers. Input is the first `first_n_chunks` chunks (default 5), truncated to 5000 characters at `overview_builder.py:37`. | +| `contextualizer.system` | `rag_system/indexing/contextualizer.py` `12` | `SYSTEM_PROMPT` | enrichment model | Role instruction for the summariser. | +| `contextualizer.local_context` | same file `14-16` | `LOCAL_CONTEXT_PROMPT_TEMPLATE` | โ€” | Wraps the neighbouring-chunk window in `` tags. | +| `contextualizer.chunk` | same file `18-26` | `CHUNK_PROMPT_TEMPLATE` | โ€” | Shows the target chunk and carries the actual instruction: a 2-5 sentence context summary, "Answer *only* with the succinct context and nothing else." | -## 4. Agent โ€“ Classical Loop +The three contextualizer parts are concatenated into a single `/api/generate` completion prompt at `contextualizer.py:42-51` (no chat roles) and sent at `contextualizer.py:53` with `enable_thinking=False`. -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `agent.loop.initial_thought` | `rag_system/agent/loop.py` `157-180` | First LLM call to think about query. | -| `agent.loop.verify_path` | same `190-205` | Secondary thought loop. | -| `agent.loop.compose_sub` | same `506-542` | Compose answer from sub-answers. | -| `agent.loop.router` | same `648-660` | Decide which subsystem handles query. | +## 2. Retrieval / query transformation -## 5. Verifier +| ID | File & lines | Model | Purpose | +|----|--------------|-------|---------| +| `query_transformer.decompose` | `rag_system/retrieval/query_transformer.py` `38-84` (system) + `87-196` (few-shot examples) + `250-265` (assembly) | utility model (`loop.py:30`) | Resolve pronouns/ellipsis against the last 5 turns, then split the query into standalone sub-queries. Returns RFC-8259 JSON `{requires_decomposition, reasoning, resolved_query, sub_queries}`; sent with `format="json"` at `:268`; the list is deduplicated and capped by `max_sub_queries` at `:290` (default 10). | +| `query_transformer.decompose_single_turn` | `rag_system/retrieval/query_transformer.py`, `QueryDecomposer._decompose_single_turn` | utility model | The byte-frozen single-turn decomposer โ€” the default path for history-less queries, and the prompt every single-turn bench number was measured against (arm C..K). It must stay byte-identical: even cosmetic edits (smart quotes, added examples) were measured to shift temp-0 decompositions on the gold set (arm L, 2026-08-16); any change re-triggers the full 5-bench gate. | -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `verifier.fact_check` | `rag_system/agent/verifier.py` `18-58` | Strict JSON-format grounding verifier. | +| `retrieval.retry.reformulate` | `rag_system/pipelines/retrieval_pipeline.py`, `_reformulate_query` | enrichment/utility model, `format="json"` | Fires **only** when the evidence-sufficiency retry triggers (roadmap 2.1). Asks for one rewrite of a weak-evidence query into the concrete nouns and synonyms a document would use, preserving every entity and constraint. Returns `{"query": "โ€ฆ"}`; `format="json"` is what keeps a small model's thinking preamble out of the rewritten query. | -## 6. Backend Router (Fast path) +Two legacy example blocks still live in the file at `199-203` and `205-248`, but they are excluded from the assembled prompt โ€” the two lines that would concatenate them are commented out at `:253-254`. -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `backend.router` | `backend/server.py` `435-448` | Decide "RAG vs direct LLM" before heavy processing. | +## 3. Answer synthesis -## 7. Miscellaneous +| ID | File & lines | Model | Purpose | +|----|--------------|-------|---------| +| `retrieval_pipeline.synth_final` | `rag_system/pipelines/retrieval_pipeline.py`, `_synthesize_final_answer` | generation model | Turn the retrieved snippets into the final answer (7 numbered hard rules; instructs the model to reply exactly "I could not find that information in the provided documents." when the snippets do not cover the question; rule 7 covers the `[Source document: โ€ฆ]` line each snippet carries). Streamed token-by-token via `stream_completion`; each token is forwarded to the `event_callback` as a `token` event. | -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `vision.placeholder` | `rag_system/utils/ollama_client.py` `169` | Dummy prompt for VLM colour check. | +## 4. Agent loop (`rag_system/agent/loop.py`) ---- +| ID | Lines | Model | Purpose | +|----|-------|-------|---------| +| `agent.loop.history_wrapper` | `163-171` | none โ€” template only | `_format_query_with_history` builds the `contextual_query` string that is embedded into downstream prompts. It makes no LLM call. | +| `agent.loop.overview_router` | `615-630` | utility model (call at `632-634`, `format="json"`) | First routing pass. Interpolates the loaded document overviews (first 40, `:612-613`) under a `DOCUMENT OVERVIEWS:` header and returns `{"category": "direct_answer"}` or `{"category": "rag_query"}`. | +| `agent.loop.triage_fallback` | `agent/loop.py`, `_triage_query_async` | utility model, `format="json"` | Last-resort routing. Reached only when the overview router returns `None` (no overviews loaded) **and** there is no chat history. Two-way vocabulary: `rag_query` / `direct_answer`; `_normalize_triage()` maps anything else to `rag_query`. | +| `agent.loop.direct_answer` | `331-336` | generation model (streamed at `342-344`) | Answers on the `direct_answer` route from conversation history or general knowledge. Caps the reply at 1-2 sentences. | +| `agent.loop.compose_sub` | `490-515` | generation model (streamed at `519-522`) | Compose one final answer from the JSON list of sub-question/sub-answer pairs. Used when `query_decomposition.compose_from_sub_answers` is true and decomposition produced more than one sub-query. | -### Missing / To-Do -1. Verify whether **ReActAgent.PROMPT_TEMPLATE** captures every placeholder โ€“ some earlier lines may need explicit ID when we move to central registry. -2. Search TS/JS code once the backend prompts are ported (currently none). +## 5. Verifier + +| ID | File & lines | Model | Purpose | +|----|--------------|-------|---------| +| `verifier.fact_check` | `rag_system/agent/verifier.py` `25-85` | utility model (`loop.py:29`) | Grounding check with three few-shot examples and a `# TASK` block. The prompt is built in four appends: the base f-string ends at `:75`, the context is appended clamped to 4000 characters at `:76`, the answer at `:81`, and the `` tag at `:82-85`. Sent asynchronously with `format="json"`. Verdict labels: `SUPPORTED` / `NOT_SUPPORTED` / `NEEDS_CLARIFICATION`. **Skipped entirely when `VERIFIER_MODEL` / `verification.model` names a local NLI verifier** โ€” that backend makes no LLM call at all (roadmap 2.4, `verifier.md`). | + +## 6. Backend router (fast path) โ€” removed + +The gateway's old "Respond with exactly one word: USE_RAG or DIRECT_LLM" LLM router prompt was deleted in Phase 2.3. Gateway routing is now the deterministic no-LLM gate `should_use_rag` (`backend/server.py`), which makes no model call at all, so there is no backend prompt to inventory. --- -**Next step:** create `rag_system/prompts/registry.yaml` and start moving each prompt above into a keyโ€“value entry with identical IDs. Update callers gradually using the helper proposed earlier. \ No newline at end of file +### Notes + +* `rag_system/utils/watsonx_client.py:222` contains the string `prompt="What is AI?"`, but it is literal text inside a `print()` usage banner โ€” it is never sent to a model and is therefore not inventoried. +* There is no prompt registry module; every prompt above is an inline literal at the cited location. +* There is no ReAct-style think/act/observe prompt anywhere. The agent's stages are triage โ†’ (optional) decomposition โ†’ retrieval โ†’ rerank โ†’ expand โ†’ prune โ†’ synthesis โ†’ verification. +* The eval judge prompts (`eval/judge.py`) exist but are out of scope of this runtime inventory. diff --git a/Documentation/quick_start.md b/Documentation/quick_start.md index 3a68f48d..b92b8a6d 100644 --- a/Documentation/quick_start.md +++ b/Documentation/quick_start.md @@ -1,34 +1,37 @@ -# โšก Quick Start Guide - RAG System +# โšก Quick Start Guide - LocalGPT -_Get up and running in 5 minutes!_ +_Get up and running in about 10 minutes (plus model download time)._ --- ## ๐Ÿš€ Choose Your Deployment Method -### Option 1: Docker Deployment (Production Ready) ๐Ÿณ +### Option 1: Docker Deployment ๐Ÿณ -Best for: Production deployments, isolated environments, easy scaling +Best for: isolated environments, running the same stack everywhere. -### Option 2: Direct Development (Developer Friendly) ๐Ÿ’ป +### Option 2: Direct Development ๐Ÿ’ป -Best for: Development, customization, debugging, faster iteration +Best for: development, customization, debugging, faster iteration. + +Both need Ollama. The Docker path runs Ollama on the host by default (better GPU +access), but can also run it as a container. --- ## ๐Ÿณ Docker Deployment ### Prerequisites -- Docker Desktop installed and running +- Docker Desktop (or Docker Engine 24+ with the Compose plugin) installed and running - 8GB+ RAM available -- Internet connection +- Internet connection (first build downloads Python and Node packages; first query + downloads the embedding model from HuggingFace; the reranker follows on first use) -### Step 1: Clone and Setup +### Step 1: Clone ```bash -# Clone repository -git clone -cd rag_system_old +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Ensure Docker is running docker version @@ -36,7 +39,7 @@ docker version ### Step 2: Install Ollama Locally -**Even with Docker, Ollama runs locally for better performance:** +**By default the containers talk to Ollama on the host:** ```bash # Install Ollama @@ -46,32 +49,53 @@ curl -fsSL https://ollama.ai/install.sh | sh ollama serve # Install models (in another terminal) -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, enrichment, verification ``` +Prefer not to install Ollama on the host? Skip this step and use +`./start-docker.sh container` below. + ### Step 3: Start Docker Containers ```bash -# Start all containers +# Start all containers against local Ollama ./start-docker.sh # Or manually: docker compose --env-file docker.env up --build -d ``` +Containerized Ollama instead: + +```bash +./start-docker.sh container +# Pull the models inside the container the first time +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` + +`./start-docker.sh` with no argument checks port 11434. If nothing is listening it +offers to switch to the containerized Ollama; pass `-y` (or set `NONINTERACTIVE=1`) +to accept that without a prompt in scripts. + ### Step 4: Verify Deployment ```bash -# Check container status +# Check container status (backend waits for rag-api to report healthy) docker compose ps # Test endpoints -curl http://localhost:3000 # Frontend -curl http://localhost:8000/health # Backend -curl http://localhost:8001/models # RAG API +curl http://localhost:3000 # Frontend +curl http://localhost:8000/health # Backend +curl http://localhost:8001/health # RAG API ``` +The `rag-api` container loads the agent and the embedding model at startup, so its +health check has a 120s start period and the first `docker compose up` can take +several minutes before `backend` starts. The reranker (~7.5 GB) is fetched lazily on +the first query that reranks โ€” expect one slow first answer. + ### Step 5: Access Application Open your browser to: **http://localhost:3000** @@ -81,21 +105,20 @@ Open your browser to: **http://localhost:3000** ## ๐Ÿ’ป Direct Development ### Prerequisites -- Python 3.8+ -- Node.js 16+ and npm +- Python 3.10+ (3.11 recommended) +- Node.js 20+ and npm - 8GB+ RAM available ### Step 1: Clone and Install Dependencies ```bash -# Clone repository -git clone -cd rag_system_old +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Install Python dependencies pip install -r requirements.txt -# Install Node.js dependencies +# Install Node.js dependencies npm install ``` @@ -109,8 +132,8 @@ curl -fsSL https://ollama.ai/install.sh | sh ollama serve # Install models (in another terminal) -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` ### Step 3: Start the System @@ -120,29 +143,39 @@ ollama pull qwen3:8b python run_system.py ``` -**Or start components manually in separate terminals:** +`run_system.py` reuses an already-running Ollama, pulls any missing model, then +starts the RAG API, the backend and the frontend. + +**Or start components manually in separate terminals โ€” all from the repository root:** ```bash # Terminal 1: RAG API python -m rag_system.api_server # Terminal 2: Backend -cd backend && python server.py +python backend/server.py # Terminal 3: Frontend npm run dev ``` +> Do not `cd backend` first. The SQLite path defaults to `backend/chat_data.db` +> relative to the working directory, so running from inside `backend/` creates a +> second database at `backend/backend/chat_data.db`. + ### Step 4: Verify Installation ```bash -# Check system health +# Loads the models and runs a sample query against the first LanceDB table python system_health_check.py +# HTTP health check per service (exits non-zero if a required one is unhealthy) +python run_system.py --health + # Test endpoints -curl http://localhost:3000 # Frontend -curl http://localhost:8000/health # Backend -curl http://localhost:8001/models # RAG API +curl http://localhost:3000 # Frontend +curl http://localhost:8000/health # Backend +curl http://localhost:8001/health # RAG API ``` ### Step 5: Access Application @@ -158,14 +191,16 @@ Open your browser to: **http://localhost:3000** - Give your session a descriptive name ### 2. Upload Documents -- Click "Create New Index" button -- Upload PDF files from your computer +- Click "Create New Index" +- Upload PDF, DOCX, TXT, MD or HTML files - Configure processing options: - - **Chunk Size**: 512 (recommended) - - **Embedding Model**: Qwen/Qwen3-Embedding-0.6B + - **Chunk Size**: 512 (default) + - **Embedding Model**: `microsoft/harrier-oss-v1-0.6b` (default) - **Enable Enrichment**: Yes - Click "Build Index" and wait for processing +Indexing is synchronous: the request stays open until the pipeline finishes. + ### 3. Start Chatting - Select your built index - Ask questions about your documents: @@ -174,6 +209,16 @@ Open your browser to: **http://localhost:3000** - "What are the main findings?" - "Compare the arguments in section 3 and 5" +The chat settings panel exposes the same knobs the API does: search type +(`hybrid` / `vector_only` / `fts_only`), retrieval_k, reranker top-k, context +window, decomposition, verification, Provence pruning, and "Stream phases". + +> **Stream phases is on by default**, and that path streams straight from the RAG +> API to the browser. When the stream completes, the UI saves the finished turn +> through the gateway (`POST /sessions/{id}/messages/save`), so the conversation +> **is** written to the chat history database. Only direct (non-UI) stream +> consumers must save the turn themselves. + --- ## ๐Ÿ”ง Management Commands @@ -182,31 +227,38 @@ Open your browser to: **http://localhost:3000** ```bash # Container management -./start-docker.sh # Start all containers -./start-docker.sh stop # Stop all containers -./start-docker.sh logs # View logs -./start-docker.sh status # Check status +./start-docker.sh # Start (local Ollama) +./start-docker.sh container # Start (containerized Ollama) +./start-docker.sh stop # Stop all containers +./start-docker.sh logs # View logs +./start-docker.sh status # Check status +./start-docker.sh help # Usage # Manual Docker Compose docker compose ps # Check status -docker compose logs -f # Follow logs -docker compose down # Stop containers -docker compose up --build -d # Rebuild and start +docker compose logs -f # Follow logs +docker compose down # Stop containers +docker compose --env-file docker.env up --build -d # Rebuild and start ``` ### Direct Development Commands ```bash # System management -python run_system.py # Start all services -python system_health_check.py # Check system health - -# Individual components -python -m rag_system.api_server # RAG API only -cd backend && python server.py # Backend only -npm run dev # Frontend only - -# Stop: Press Ctrl+C in terminal running services +python run_system.py # Start all services +python run_system.py --mode prod # `npm run build` then `next start` +python run_system.py --no-frontend # Ollama + RAG API + backend only +python run_system.py --health # HTTP health checks +python run_system.py --logs-only # Tail logs/*.log from another shell +python run_system.py --stop # Stop everything in logs/run_system.pid +python system_health_check.py # Deep check: models, LanceDB, sample query + +# Individual components (from the repository root) +python -m rag_system.api_server # RAG API only +python backend/server.py # Backend only +npm run dev # Frontend only + +# Stop: Ctrl+C in the terminal running the services, or `python run_system.py --stop` ``` --- @@ -220,8 +272,9 @@ npm run dev # Frontend only # Check Docker daemon docker version -# Restart Docker Desktop and try again -./start-docker.sh +# The backend will not start until rag-api reports healthy +docker compose ps +docker compose logs -f rag-api ``` **Port conflicts?** @@ -238,7 +291,7 @@ lsof -i :3000 -i :8000 -i :8001 **Import errors?** ```bash # Check Python installation -python --version # Should be 3.8+ +python --version # 3.10+ required, 3.11 recommended # Reinstall dependencies pip install -r requirements.txt --force-reinstall @@ -247,7 +300,7 @@ pip install -r requirements.txt --force-reinstall **Node.js errors?** ```bash # Check Node version -node --version # Should be 16+ +node --version # Should be 20+ # Reinstall dependencies rm -rf node_modules package-lock.json @@ -272,20 +325,25 @@ ollama serve docker stats # For Docker htop # For direct development -# Recommended: 16GB+ RAM for optimal performance +# Use lighter models +export GENERATION_MODEL=qwen3.5:4b +# (the default embedder is already the small one, ~1.2GB) ``` +**Answers ignore my documents?** +Check that the session is linked to a built index โ€” the backend only routes to the +RAG API when the session has one. Sending `"force_rag": true` on +`POST /sessions/{id}/messages` bypasses the router. + --- ## ๐Ÿ“Š System Verification -Run this comprehensive check: - ```bash # Check all endpoints curl -f http://localhost:3000 && echo "โœ… Frontend OK" -curl -f http://localhost:8000/health && echo "โœ… Backend OK" -curl -f http://localhost:8001/models && echo "โœ… RAG API OK" +curl -f http://localhost:8000/health && echo "โœ… Backend OK" +curl -f http://localhost:8001/health && echo "โœ… RAG API OK" curl -f http://localhost:11434/api/tags && echo "โœ… Ollama OK" # For Docker: Check containers @@ -298,82 +356,116 @@ docker compose ps If you see: - โœ… All services responding -- โœ… Frontend accessible at http://localhost:3000 +- โœ… Frontend accessible at http://localhost:3000 - โœ… No error messages You're ready to start using LocalGPT! ### What's Next? -1. **๐Ÿ“š Upload Documents**: Add your PDF files to create indexes +1. **๐Ÿ“š Upload Documents**: Add files to create an index 2. **๐Ÿ’ฌ Start Chatting**: Ask questions about your documents -3. **๐Ÿ”ง Customize**: Explore different models and settings -4. **๐Ÿ“– Learn More**: Check the full documentation below +3. **๐Ÿ”ง Customize**: Try `RAG_CONFIG_MODE=fast`, different models, other retrieval modes +4. **๐Ÿ“– Learn More**: Check the documentation below ### ๐Ÿ“ Key Files ``` -rag-system/ +localGPT/ โ”œโ”€โ”€ ๐Ÿณ start-docker.sh # Docker deployment script โ”œโ”€โ”€ ๐Ÿƒ run_system.py # Direct development launcher -โ”œโ”€โ”€ ๐Ÿฉบ system_health_check.py # System verification +โ”œโ”€โ”€ ๐Ÿฉบ system_health_check.py # Deep system verification +โ”œโ”€โ”€ ๐Ÿ› ๏ธ create_index_script.py # Interactive / batch index creation โ”œโ”€โ”€ ๐Ÿ“‹ requirements.txt # Python dependencies โ”œโ”€โ”€ ๐Ÿ“ฆ package.json # Node.js dependencies +โ”œโ”€โ”€ โš™๏ธ .env.example # Every environment variable with its default โ”œโ”€โ”€ ๐Ÿ“ Documentation/ # Complete documentation -โ””โ”€โ”€ ๐Ÿ“ rag_system/ # Core system code +โ”œโ”€โ”€ ๐Ÿ“ rag_system/ # RAG API, agent, pipelines (config in main.py) +โ”œโ”€โ”€ ๐Ÿ“ backend/ # Gateway server + SQLite database +โ””โ”€โ”€ ๐Ÿ“ src/ # Next.js frontend ``` ### ๐Ÿ“– Additional Resources - **๐Ÿ—๏ธ Architecture**: See `Documentation/architecture_overview.md` -- **๐Ÿ”ง Configuration**: See `Documentation/system_overview.md` +- **๐Ÿ”ง Configuration**: See `Documentation/system_overview.md` - **๐Ÿš€ Deployment**: See `Documentation/deployment_guide.md` +- **๐Ÿณ Docker**: See `Documentation/docker_usage.md` - **๐Ÿ› Troubleshooting**: See `DOCKER_TROUBLESHOOTING.md` --- -**Happy RAG-ing! ๐Ÿš€** - ---- +## ๐Ÿ› ๏ธ Indexing Without the UI -## ๐Ÿ› ๏ธ Indexing Scripts - -The repository includes several convenient scripts for document indexing: - -### Simple Index Creation Script - -For quick document indexing without the UI: +### Built-in CLI ```bash -# Basic usage -./simple_create_index.sh "Index Name" "document.pdf" +# Index a file or a whole directory with the 'default' profile +python -m rag_system.main index ./my_documents + +# Speed-optimised profile +python -m rag_system.main index ./my_documents --mode fast -# Multiple documents -./simple_create_index.sh "Research Papers" "paper1.pdf" "paper2.pdf" "notes.txt" +# Ask one question and print the JSON result +python -m rag_system.main chat "What are the key findings?" --mode default -# Using wildcards -./simple_create_index.sh "Invoice Collection" ./invoices/*.pdf +# Start the RAG API (same as `python -m rag_system.api_server`) +python -m rag_system.main api --port 8001 ``` -**Supported file types**: PDF, TXT, DOCX, MD +`index` walks a directory for `.pdf`, `.docx`, `.html`, `.htm`, `.md` and `.txt`. +It writes into the profile's shared table (`text_pages_v4`), not into a per-index +table, so indexes built this way are not listed in the web UI. -### Batch Indexing Script +### Interactive / Batch Script -For processing large document collections: +For an index the web UI can see, use `create_index_script.py` โ€” it creates the +database row, uploads the document records and writes to `text_pages_`: ```bash -# Using the Python batch indexing script -python demo_batch_indexing.py - -# Or using the direct indexing script +# Guided prompts python create_index_script.py + +# Write a template, edit it, then run it +python create_index_script.py --create-sample # writes index_config.sample.json +python create_index_script.py --batch index_config.sample.json + +# Use a custom pipeline config instead of PIPELINE_CONFIGS["default"] +python create_index_script.py --config my_pipeline.json +``` + +The batch file looks like this โ€” replace the placeholder paths with your own +absolute paths: + +```json +{ + "index_name": "Sample Batch Index", + "index_description": "Example batch index configuration", + "documents": [ + "/absolute/path/to/first.pdf", + "/absolute/path/to/second.pdf" + ], + "processing": { + "chunk_size": 512, + "enable_enrich": true, + "enable_latechunk": true, + "enable_docling": true, + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "retrieval_mode": "hybrid", + "window_size": 2 + } +} ``` -These scripts automatically: -- โœ… Check prerequisites (Ollama running, Python dependencies) -- โœ… Validate document formats -- โœ… Create database entries -- โœ… Process documents with the RAG pipeline -- โœ… Generate searchable indexes +Both paths: +- โœ… Parse documents with Docling (OCR fallback for scanned PDFs) +- โœ… Chunk, optionally enrich, and embed +- โœ… Write vectors plus a native full-text index into LanceDB +- โœ… Generate a document overview used by the query router + +The script exits non-zero on failure and deletes the half-created index row. + +--- ---- \ No newline at end of file +**Happy RAG-ing! ๐Ÿš€** diff --git a/Documentation/research/README.md b/Documentation/research/README.md new file mode 100644 index 00000000..d75b633a --- /dev/null +++ b/Documentation/research/README.md @@ -0,0 +1,30 @@ +# Research Evidence (August 2026) + +Three independent research sweeps on the state of the art in agentic retrieval, +compiled 2026-08-08 by LLM research agents from primary sources only (first-party +engineering blogs, arXiv/ACL/ICLR/SIGIR papers, official model cards and +leaderboards). Anonymous blog posts and SEO content were explicitly rejected; +each report carries its own rejection log and a list of claims that could NOT +be verified. + +| File | Scope | +|------|-------| +| [industry-evidence-2026.md](industry-evidence-2026.md) | What Anthropic, OpenAI, Google, Microsoft, LlamaIndex, LangChain, Weaviate, Qdrant, Vespa, Pinecone, Elastic, Exa, Perplexity, Glean et al. have published about production agentic retrieval โ€” architectures, numbers, and where vendors directly contradict each other | +| [academic-evidence-2026.md](academic-evidence-2026.md) | The 2025โ€“26 literature: RL-trained search agents, reasoning-aware retrieval (BRIGHT lineage), late interaction, hybrid fusion, late chunking, context compression, verification/attribution, GraphRAG, memory โ€” with the controlled replications that deflated the 2024 headline claims | +| [component-map-2026.md](component-map-2026.md) | Component-by-component SOTA map for local/self-hosted stacks: parsing, chunking, embeddings, sparse+fusion, rerankers, query planning, routing, loop patterns, verification, compression, memory, evaluation โ€” each claim graded established / emerging / contested | + +## How to read these + +- These are **evidence documents, not descriptions of localGPT**. Nothing in + them implies a feature exists in this repo. What localGPT actually does is + documented in the rest of `Documentation/`. +- Every claim carries its source and date. Claims graded *established* are + peer-reviewed or independently replicated; *vendor* numbers are self-reported + and directionally credible at best. +- The actionable distillation lives in + [../research_roadmap.md](../research_roadmap.md), which maps this evidence to + concrete, staged changes with acceptance criteria. + +Provenance: generated with web access on 2026-08-08; model knowledge cutoffs +predate several cited releases, so everything recent is web-sourced. Each +report's own "could not verify / do not cite" appendix applies. diff --git a/Documentation/research/academic-evidence-2026.md b/Documentation/research/academic-evidence-2026.md new file mode 100644 index 00000000..de06bcc2 --- /dev/null +++ b/Documentation/research/academic-evidence-2026.md @@ -0,0 +1,811 @@ +All four research streams are complete. Here is the deliverable. + +--- + +# ACADEMIC STATE OF THE ART IN AGENTIC RETRIEVAL โ€” 2025โ€“2026 + +**Method / verification status.** ~200 web searches (session budget exhausted) plus ~180 direct fetches of arXiv abs/HTML pages, ACL Anthology, ICLR/OpenReview, the official BRIGHT leaderboard, the Vectara leaderboard repo, and the arXiv metadata API. Every arXiv ID below was either fetched directly or returned by arXiv's own API. Items I could not verify are flagged โš ๏ธ inline and listed again at the end. Citation counts come from the Semantic Scholar API (retrieved 8 Aug 2026); OpenAlex was tried first and badly undercounts preprints, so I discarded it. + +--- + +## 1. AGENTIC / ITERATIVE RETRIEVAL TRAINING + +### 1.1 The 2025 founding wave โ€” what each actually claimed + +| System | Paper | Date | Venue | Core claim | +|---|---|---|---|---| +| **DeepRetrieval** | Pengcheng Jiang, Jiacheng Lin, Lang Cao, Runchu Tian, SeongKu Kang, Zifeng Wang, Jimeng Sun, Jiawei Han. arXiv:2503.00223 | 28 Feb 2025 (v3 12 Apr) | arXiv only, cs.IR | GRPO-trained **one-shot query generation**, reward = retrieval metric, no supervised reference queries | +| **R1-Searcher** | Huatong Song, Jinhao Jiang, Yingqian Min, Jie Chen, Zhipeng Chen, Wayne Xin Zhao, Lei Fang, Ji-Rong Wen (RUC). arXiv:2503.05592 | 7 Mar 2025 | arXiv | **Two-stage outcome-based RL**, no process rewards, no distillation cold start | +| **Search-R1** | Bowen Jin, Hansi Zeng, Zhenrui Yue, Jinsung Yoon, Sercan Arik, Dong Wang, Hamed Zamani, Jiawei Han (UIUC + UMass + Google Cloud AI). arXiv:2503.09516, v5 5 Aug 2025 | 12 Mar 2025 | arXiv (**1,317 citations**, 5.3k GitHub stars) | Multi-turn interleaved search+reason, **retrieved-token masking**, outcome-only reward | +| **ReSearch** | Mingyang Chen et al. arXiv:2503.19470 | 25 Mar 2025 | arXiv | End-to-end RL, search treated as part of the reasoning chain, no supervised tool-use trajectories | +| **WebDancer** | Alibaba Tongyi Lab. arXiv (WebAgent family) | May 2025 | **NeurIPS 2025** | 4-stage: data construction โ†’ trajectory sampling โ†’ SFT โ†’ RL | +| **WebSailor** | Kuan Li, Zhongwang Zhang, Huifeng Yin, โ€ฆ Yong Jiang, Ming Yan, Pengjun Xie, Fei Huang, Jingren Zhou (Alibaba Tongyi). arXiv:2507.02592 | 3 Jul 2025 | arXiv | High-uncertainty synthetic tasks + RFT cold start + **DUPO** RL | +| **ZeroSearch** | Hao Sun, Zile Qiao, โ€ฆ Fei Huang, Jingren Zhou (Alibaba). arXiv:2505.04588, v3 19 May 2026 | 7 May 2025 | arXiv | Replace the live search engine with a **simulated LLM retriever** during RL | + +**Search-R1's actual numbers (verified from v5 HTML, not the abstract).** Setup: 2018 Wikipedia dump, **E5 retriever, top-3 passages**, PPO (more stable) vs GRPO (faster convergence but reward collapse). Average EM across NQ, TriviaQA, PopQA, HotpotQA, 2Wiki, Musique, Bamboogle: + +| Method | Qwen2.5-7B avg EM | Qwen2.5-3B avg EM | +|---|---|---| +| Direct | 0.181 | 0.134 | +| CoT | 0.106 | 0.015 | +| IRCoT | 0.239 | 0.181 | +| RAG | 0.304 | 0.270 | +| R1 (reason, no search) | 0.276 | 0.229 | +| **Search-R1-base** | **0.431** | **0.303** | + +The widely-quoted "**41% improvement**" is over the RAG baseline. Per-dataset 7B: NQ 0.480, TriviaQA 0.638, PopQA 0.457, HotpotQA 0.433, 2Wiki 0.382, Musique **0.196**, Bamboogle 0.432. + +โš ๏ธ **Caveats that matter.** (a) The baseline is a naive top-3 single-shot RAG over a 2018 Wikipedia dump โ€” a weak reference point by 2026 standards. (b) Musique at 0.196 shows the hard multi-hop case is barely moved. (c) The whole evaluation runs against a **fixed local E5 index**, so nothing is learned about live-web behavior. This last point is the wedge for ยง1.3. + +**DeepRetrieval's numbers:** publication search recall **65.07%** vs 24.68% prior SOTA; trial search recall **63.18%** vs 32.11%; beats GPT-4o and Claude-3.5-Sonnet on 11 of 13 datasets with a **3B** model. โš ๏ธ These are literature-search domains (PubMed/ClinicalTrials.gov), not general QA; the "hacking real search engines" framing is doing a lot of work โ€” the gains are largely query-formulation gains against a fixed API. + +### 1.2 The deep-research agent line and where the open frontier sits in mid-2026 + +**Tongyi DeepResearch Technical Report** (56 authors, Alibaba Tongyi). arXiv:2510.24701, 28 Oct 2025, v3 18 May 2026. 30.5B total / **3.3B activated** MoE, 128K context, agentic mid-training + agentic post-training. Verified Table 1: + +| Model | HLE | BrowseComp | BrowseComp-ZH | GAIA | xbench-DS | WebWalker | FRAMES | +|---|---|---|---|---|---|---|---| +| **Tongyi DeepResearch 30B-A3B** | **32.9** | 43.4 | 46.7 | **70.9** | **75.0** | **72.2** | **90.6** | +| OpenAI DeepResearch | 26.6 | **51.5** | 42.9 | 67.4 | โ€” | โ€” | โ€” | +| OpenAI o3 (ReAct) | 24.9 | 49.7 | **58.1** | โ€” | 67.0 | 71.7 | 84.0 | +| DeepSeek-V3.1 (ReAct) | 29.8 | 30.0 | 49.2 | 63.1 | 71.0 | 61.2 | 83.7 | +| Claude-4-Sonnet (ReAct) | 20.3 | 12.2 | 29.1 | 68.3 | 65.0 | 61.7 | 80.7 | +| GLM-4.5 (ReAct) | 21.2 | 26.4 | 37.5 | 66.0 | 70.0 | 65.6 | 78.9 | +| Kimi Researcher | 26.9 | โ€” | โ€” | โ€” | 69.0 | โ€” | 78.8 | + +Note the split verdict: a trained 30B-A3B open model **beats OpenAI Deep Research on HLE (32.9 vs 26.6) and GAIA (70.9 vs 67.4) but loses on BrowseComp-EN (43.4 vs 51.5)**. Also note Claude-4-Sonnet's 12.2 on BrowseComp under a plain ReAct harness โ€” prompted frontier models without a deep-research scaffold are not competitive on this task class. + +**What superseded them in 2026:** + +- **LiteResearcher** (Bince Qu, Wanli Li, Bo Pan, Jianyu Zhang, Zheng Liu, Pan Zhang, Wei Chen, Bo Zhang). arXiv:2604.17931, 20 Apr 2026, v5 26 Jul 2026. Builds a **"lite virtual world"** mirroring real search dynamics so RL doesn't depend on a live search API. LiteResearcher-**4B**: GAIA-Text 71.3, xbench-DS 78.0, FRAMES 83.1, WebWalker 72.7, Seal-0 41.8, HLE 22.0, BrowseComp 27.5, BrowseComp-ZH 32.5. **Parity with Claude-4.5-Sonnet on GAIA (71.3 vs 71.2) and beats Tongyi-30B on GAIA and xbench, at 4B.** Ablation (their Table 9): **SFT alone 55.58 GAIA โ†’ RL 71.3 (+15.7); xbench 64.25 โ†’ 78.0 (+13.8).** Stated limitations: 128K context exhausts on deep BrowseComp chains; gains depend on a 32M-page enriched corpus and generalization beyond it is unexplored. + +- **OpenSeeker-v2** (Yuwen Du, Rui Ye, Shuo Tang, Keduan Huang, Xinyu Zhu, Yuzhu Cai, Siheng Chen). arXiv:2605.04036, 5 May 2026. **SFT only, no RL**, 10.6k trajectories: **BrowseComp 46.0, BrowseComp-ZH 58.1, HLE 34.6, xbench 78.0** at 30B/ReAct. Self-described as the first SOTA search agent at its scale from a purely academic team using only SFT. + +- **WebExplorer** (Junteng Liu et al.). arXiv:2509.06501, 8 Sep 2025. **WebExplorer-8B**, 128K, up to 100 tool turns, averages **16 search turns after RL**; higher BrowseComp-en/zh than **WebSailor-72B** and best among โ‰ค100B on WebWalkerQA and FRAMES. + +**โ†’ OPEN DEBATE #1 โ€” is RL actually necessary?** LiteResearcher measures RL contributing **+15.7 GAIA points over its own SFT checkpoint**. OpenSeeker-v2, one month later, reaches **higher BrowseComp (46.0 vs 27.5) and HLE (34.6 vs 22.0) with pure SFT on 10.6k curated high-difficulty trajectories**. Different model scales (4B vs 30B) so it is not a clean head-to-head, but the field has no controlled experiment isolating RL from trajectory-data quality at fixed scale. This is unresolved and under-discussed. + +### 1.3 The critical literature โ€” this is the part that changes conclusions + +**โญ BrowseComp-Plus: A Fair and Disentangled Evaluation Benchmark for Deep Search Agents.** Zijian Chen, Xueguang Ma, Shengyao Zhuang, Ping Nie, Kai Zou, Andrew Liu, Joshua Green, Kshama Patel, Ruoxi Meng, Mingyi Su, Sahel Sharifymoghaddam, Yanxi Li, Haoran Hong, Xinyu Shi, Xuye Liu, Nandan Thakur, Crystina Zhang, Luyu Gao, Wenhu Chen, **Jimmy Lin** (Waterloo + CSIRO + CMU + Queensland). arXiv:2508.06600, 8 Aug 2025. **ACL 2026 Main** (aclanthology.org/2026.acl-long.1023). 155 citations. + +Fixed curated corpus, human-verified supporting documents, mined hard negatives โ€” so retriever and agent can be varied independently. Verified Table 1 accuracy: + +| Agent | + BM25 | + Qwen3-Embed-8B | +|---|---|---| +| **GPT-5** | 55.90% | **70.12%** | +| o3 | 49.28% | 63.49% | +| gpt-oss-120B-high | 28.67% | 42.89% | +| Gemini 2.5 Pro | 19.04% | 28.67% | +| Claude Opus 4 | 15.54% | 36.14% | +| Claude Sonnet 4 | 14.34% | 36.75% | +| gpt-4.1 | 14.58% | 35.42% | +| **Search-R1-32B** | **3.86%** | **10.36%** | +| Qwen3-32B | 3.49% | 10.36% | + +Four findings, each load-bearing: + +1. **Retriever quality outweighs agent choice.** Swapping BM25 โ†’ Qwen3-Embedding-8B buys GPT-5 **+14.2 points** and roughly doubles weaker agents. Better retrievers also **reduce** search-call count. +2. **RL-trained open agents do not transfer.** Search-R1-32B scores **exactly what its untrained base Qwen3-32B scores (10.36% with the good retriever)**. The RL training bought nothing outside its training distribution. This is the single most damaging result for the Search-R1 lineage's generalization claims. +3. **The bottleneck is interleaved tool-use reasoning, not knowledge.** Authors: open models "do not substantially lag behind proprietary models in their ability to answer questions when provided with sufficient evidence." Proprietary models average **20+ search calls/query**; open models fewer than 2 despite explicit tool prompting. +4. **Reasoning-specialized retrievers underperform scaled general ones inside agentic loops.** Qwen3-Embedding-8B: **14.5% Recall@5, 20.3 nDCG@10**; ReasonIR-8B: **12.2% / 16.8**. Note the absolute ceiling โ€” the *best* retriever gets 20.3 nDCG@10. Enormous headroom. + +**โญ Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?** Yang Yue, Zhiqi Chen, Rui Lu, Andrew Zhao, Zhaokai Wang, Shiji Song, Gao Huang (Tsinghua). arXiv:2504.13837, 18 Apr 2025, v5 24 Nov 2025. **NeurIPS 2025 Oral**; ICML 2025 AI4MATH workshop best paper. RLVR-trained models beat base models at pass@1 but **base models overtake at large k** โ€” the reasoning was already in the base model; RLVR narrows the sampling distribution. Six RLVR algorithms are "far from optimal in leveraging the potential of the base model." Not search-specific, but it is the theoretical frame for interpreting every "+41% over RAG" claim in this section. + +**โญ Demystifying deep search / WebDetective.** Maojia Song, Renhang Liu, Xinyu Wang, Yong Jiang, Pengjun Xie, Fei Huang, Jingren Zhou, Dorien Herremans, Soujanya Poria. arXiv:2510.05137, 1 Oct 2025 (v3 10 Dec 2025). Two indictments of current practice: **most benchmarks leak the reasoning path in the question text**, and single-pass-rate scoring collapses distinct failure modes. Hint-free multi-hop questions in a traceable Wikipedia sandbox, **25 SOTA models**: systematic failure at knowledge utilization *despite sufficient evidence*, and near-absent appropriate refusal when evidence is missing. "Today's systems excel at executing given reasoning paths but fail when required to discover them." Their EvidenceLoop workflow (verification loops + systematic evidence tracking) is the proposed fix. + +**โญ How to Train Your Deep Research Agent? Prompt, Reward, and Policy Optimization in Search-R1.** Yinuo Xu, Shuo Lu, Jianjie Cheng, Meng Wang, Qianlong Xie, Xingxing Wang, Ran He, Jian Liang. arXiv:2602.19526, 23 Feb 2026. Decoupled ablation of the three axes. Findings: the **"Fast Thinking" prompt template is more stable and better-performing than the Slow Thinking template used in prior work**; F1 rewards cause training collapse via answer avoidance (fixed by action-level penalties); **REINFORCE beats PPO with fewer search actions, and GRPO is the least stable**. Search-R1++ moves Qwen2.5-7B 0.403 โ†’ 0.442 and 3B 0.289 โ†’ 0.331. Read this alongside Search-R1's own PPO-over-GRPO finding โ€” the field's optimizer consensus is unsettled. + +### 1.4 The 2026 research frontier has moved to efficiency and context, not capability + +Outcome-only RL **provably induces over-search**: the ratio of no-search trajectories drops toward zero while redundant-search ratio rises. Four verified responses: + +- **HiPRAG** (Peilin Wu, Mian Zhang, Kun Wan, Wentian Zhao, Kaiyu He, Xinya Du, Zhiyu Chen). arXiv:2510.07794, v2 11 Apr 2026. **Accepted ICLR 2026.** Hierarchical process rewards over decomposed reasoning steps: average accuracy **65.4% (3B) / 67.2% (7B)** across 7 QA benchmarks with **over-search rate driven to 2.3%**. +- **SAAS** (Yunbo Tang, Chengyi Yang, Shiyu Liu, Zhishang Xiang, Zerui Chen, Qinggang Zhang, Jinsong Su). arXiv:2605.29796, 28 May 2026. Contrasts search-disabled vs search-enabled rollouts to model the knowledge boundary; boundary-aware trajectory penalties; stage-wise curriculum to avoid reward hacking. โš ๏ธ Abstract carries no numbers. +- **AutoSearch** (Jingbo Sun et al., incl. Dongbin Zhao). arXiv:2604.17337, 19 Apr 2026. Self-generated intermediate answers identify a **minimal sufficient search depth**; reward attainment, penalize over-search. โš ๏ธ No numbers in abstract. +- **FoldAct** (Jiaqi Shao, Yufeng Miao, Wei Zhang, Bing Luo). arXiv:2512.22733, 28 Dec 2025. Context folding for long-horizon RL: identifies gradient dilution on summary tokens, self-conditioning instability, and per-turn context recomputation. **5.19ร— training speedup.** +- **Erase to Improve (ERL)** (Ziliang Wang et al.). arXiv:2510.00861, v2 20 Apr 2026. Identify โ†’ erase โ†’ regenerate faulty reasoning steps. **3B: +8.48% EM / +11.56% F1; 7B: +5.38% EM / +7.22% F1** on HotpotQA/MuSiQue/2Wiki/Bamboogle. +- **Agentic-R** (Wenhan Liu, Xinyu Ma, Yutao Zhu, Yuchen Li, Daiting Shi, Dawei Yin, Zhicheng Dou). arXiv:2601.11888, 17 Jan 2026. **Bidirectional iterative co-optimization of the search agent and the retriever**, with passage utility measured by both local relevance and global answer correctness. Directionally the most interesting 2026 idea โ€” it treats the BrowseComp-Plus finding (retriever dominates) as a training target. โš ๏ธ No numbers in abstract. + +**Surveys.** "A Comprehensive Survey on Reinforcement Learning-based Agentic Search" โ€” Minhua Lin, Zongyu Wu, Zhichao Xu, Hui Liu, Xianfeng Tang, Qi He, Charu Aggarwal, Hui Liu, Xiang Zhang, Suhang Wang (Penn State + Amazon + IBM). arXiv:2510.16724, 19 Oct 2025, 38pp. Also "The Landscape of Agentic Reinforcement Learning for LLMs" (arXiv:2509.02547) and "RL Foundations for Deep Research Systems: A Survey" (arXiv:2509.06733). + +**Honest bottom line for ยง1.** RL-trained search behavior demonstrably beats prompted loops **within a fixed base model on the training distribution** (Search-R1 0.431 vs 0.304 RAG; LiteResearcher +15.7 GAIA over its own SFT). It demonstrably **does not transfer** to a new corpus/retriever (Search-R1-32B = its base model on BrowseComp-Plus). The 2026 SOTA on hard deep-research benchmarks belongs to large-scale trajectory curation (SFT or RL) plus a strong retriever โ€” and **retriever choice moves accuracy more than agent choice does**. + +--- + +## 2. REASONING-AWARE RETRIEVAL + +### 2.1 BRIGHT and what happened to it + +**BRIGHT: A Realistic and Challenging Benchmark for Reasoning-Intensive Retrieval.** Princeton et al. arXiv:2407.12883, 16 Jul 2024. **ICLR 2025** (OpenReview ykuc5q381b). 177 citations. 1,384โ€“1,398 real-world queries, 12 domains (economics, psychology, math, coding, robotics, StackExchange splits, theorem retrieval), corpora 7.9Kโ€“414K docs. + +At release: **max nDCG@10 = 24.3**. SFR-Embedding-Mistral, then MTEB #1 at 59.0, scored **18.3**. LLM reasoning-step query augmentation helped but stayed under 30. Authors report BRIGHT is robust to data leakage โ€” fine-tuning on the retrieval documents barely moves scores. + +**The official leaderboard as of 8 Aug 2026** (brightbenchmark.github.io), short-document track, avg nDCG@10 over 12 datasets: + +| Rank | System | Score | Date | Reranker | +|---|---|---|---|---| +| 1 | Mira-Reasoning-Retrieval (Forward AI Labs) | **66.9** | 22 Apr 2026 | Yes | +| 2 | INF-X-Retriever | 63.4 | 20 Dec 2025 | Yes | +| 3 | RakanEmbed4B | 52.4 | 20 Mar 2026 | Yes | +| 4 | NeMo Retriever Agentic Retrieval (NVIDIA) | 50.9 | 13 Mar 2026 | Yes | +| 5 | DIVER-v3-GroupRank | 46.8 | 13 Nov 2025 | Yes | +| 6 | BGE-Reasoner-0928 (BAAI) | 46.4 | 13 Oct 2025 | Yes | +| 7 | Lattice Hierarchical Retrieval (Google) | 42.1 | 17 Oct 2025 | Yes | +| โ€” | BM25 + GPT-4 reasoning + reranking | 30.4 | โ€” | Yes | +| โ€” | **BM25 alone** | **14.5** | โ€” | No | + +Long-document track (avg Recall@1, 8 datasets, docs up to ~40k): BM25 = 11.4. + +**โš ๏ธ Four caveats that reframe the 66.9.** +1. **Every single top entry uses a reranker.** BRIGHT in 2026 measures *pipelines with heavy test-time compute*, not retrievers. +2. **The gain is mostly LLM query expansion, not retrieval.** DIVER (Duolin Sun et al., Ant Group, arXiv:2508.07995, v5 2 Apr 2026) reports **46.8 overall but 31.9 on original queries** โ€” a 14.9-point gap attributable to iterative LLM query rewriting. BM25 alone 14.5 โ†’ BM25 + GPT-4 reasoning + rerank 30.4 makes the same point: **LLM expansion roughly doubles a 1994 lexical baseline**. +3. **Verification varies by entry.** Ranks 5โ€“7 link arXiv papers or GitHub; ranks 1โ€“3 link personal/company webpages. Treat the 66.9 as an unrefereed submission. +4. Compare to BrowseComp-Plus, where the best retriever inside an agentic loop reaches **20.3 nDCG@10**. Standalone BRIGHT scores and agentic-loop utility are not the same quantity. + +**โญ The reproducibility audit: Lighting the Way for BRIGHT.** Sahel Sharifymoghaddam, Yijun Ge, **Jimmy Lin**. arXiv:2509.02558, 2 Sep 2025, **v2 1 Jun 2026, SIGIR 2026 Reproducibility Track**. Findings: (a) BRIGHT's published baseline silently uses **query-side BM25 ("BM25Q")**, an undocumented detail that consistently outperforms standard BM25 on long queries โ€” meaning much of the literature has been comparing against a mis-specified lexical baseline; (b) **BM25Q's advantage is largely BRIGHT-specific** and does not carry to five other benchmarks, while fusion with standard BM25 does; (c) an audit of the BRIGHT corpus **uncovers data-quality issues that affect evaluation**. + +**Successors.** +- **BRIGHT-Pro** โ€” "Rethinking Reasoning-Intensive Retrieval: Evaluating and Advancing Retrievers in Agentic Search Systems." Yilun Zhao, Jinbiao Wei, Tingyu Song, Siyue Zhang, Chen Zhao, Arman Cohan. arXiv:2605.04018, 5 May 2026, **ACL 2026**. Direct critique: "benchmarks such as BRIGHT provide narrow gold sets and evaluate retrievers **in isolation**, while synthetic training corpora optimize single-passage relevance rather than **evidence portfolio construction**." Expert-annotated multi-aspect gold evidence; evaluates under **both static and agentic protocols**; introduces RTriever-Synth (aspect-decomposed, positive-conditioned hard negatives) and RTriever-4B (LoRA on Qwen3-Embedding-4B). Key claim: **agentic evaluation exposes retriever behaviors hidden by standard metrics.** +- **MM-BRIGHT** โ€” Abdelrahman Abdallah et al. arXiv:2601.09562, 14 Jan 2026. 2,803 queries, 29 technical domains, four task types. BM25 text-only **8.5** nDCG@10; best text-only (DiVeR) **32.2**; Nomic-Vision multimodal-to-text **27.6**. + +### 2.2 Reasoning-augmented retrievers and rerankers + +**ReasonIR-8B** โ€” Meta FAIR (facebookresearch/ReasonIR). arXiv:2504.20595, 29 Apr 2025. 76 citations. Synthetic pipeline generating reasoning-requiring queries plus **plausibly-related-but-unhelpful hard negatives**. **29.9 nDCG@10 on BRIGHT without reranker, 36.9 with.** Downstream: **+6.4% MMLU, +22.6% GPQA** over closed-book, beating other retrievers and search engines. + +**RaDeR** โ€” Debrup Das, Sam O'Nuallain, Razieh Rahimi. arXiv:2505.18405, 23 May 2025. Trained from **retrieval-augmented math reasoning trajectories** with self-reflective relevance evaluation; generalizes to BRIGHT and RAR-b. Two notable claims: it is the **first dense retriever to outperform BM25 when queries are chain-of-thought reasoning steps** (an admission of how bad the prior state was), and it matches/beats ReasonIR using **2.5% of ReasonIR's training data**. + +**Reason-ModernColBERT** (LightOn, 149M params). **โš ๏ธ Blog + HuggingFace model card only โ€” no arXiv paper.** The marketing says it "outperforms all models up to 7B on BRIGHT" and beats ReasonIR-8B "by more than 2.5 nDCG on average." Its own model-card table says otherwise: + +| Split group | Reason-ModernColBERT (149M) | ReasonIR-8B | BM25 | +|---|---|---|---| +| Mean StackExchange | **27.43** | 24.76 | 17.21 | +| Mean Coding | 19.79 | **22.75** | 16.15 | +| Mean Theorem | 15.38 | **24.60** | 7.17 | +| **Full BRIGHT mean** | 22.62 | **24.38** | 14.53 | + +**The +2.5 claim holds only on StackExchange. On the full BRIGHT mean it loses, 22.62 vs 24.38, and loses badly on Theorem.** The "45ร— smaller" framing is real; the "beats it" framing is split-selective. License is cc-by-nc-4.0 (training-data restriction), so it is not commercially usable either. + +**ReasonRank** โ€” Wenhan Liu, Xinyu Ma, Weiwei Sun, Yutao Zhu, Yuchen Li, Dawei Yin, Zhicheng Dou (RUC + Baidu). arXiv:2508.07050, 9 Aug 2025, **ACL 2026 Main**. DeepSeek-R1-synthesized reasoning-intensive ranking labels; cold-start SFT then RL with a **multi-view ranking reward** for the multi-turn nature of listwise ranking. Outperforms baselines with **lower latency than pointwise rerankers**. Was #2 on BRIGHT (40.8 reranking RaDeR, Aug 2025). + +**BGE-Reasoner** (BAAI + USTC, VectorSpaceLab/agentic-search). Multiple rewritten queries + ensembled reranking across model sizes. **46.4 nDCG@10 (BGE-Reasoner-0928, Oct 2025)**, held BRIGHT #1 briefly. โš ๏ธ GitHub/model-card, no verified paper. + +### 2.3 Test-time compute for retrieval โ€” the most interesting thread + +**Rank1** โ€” Orion Weller, Kathryn Ricci, Eugene Yang, Andrew Yates, Dawn Lawrie, Benjamin Van Durme (JHU HLTCOE). arXiv:2502.18418, 25 Feb 2025, **CoLM 2025**, 72 citations. First reranker trained to exploit test-time compute; **600,000+ R1 reasoning traces over MS MARCO** open-sourced. SOTA on reasoning and instruction-following ranking, "works remarkably well out of distribution." + +**โญ LATTICE: LLM-guided Hierarchical Search for End-to-end Reasoning Intensive Retrieval.** Nilesh Gupta, Wei-Cheng Chang, Ngot Bui, Cho-Jui Hsieh, Inderjit S. Dhillon (Google). arXiv:2510.13217, 15 Oct 2025, v2 25 May 2026. **Base LATTICE with a single off-the-shelf LLM reaches 46.7 nDCG@10 on BRIGHT โ€” matching the best fine-tuned ensemble baseline overall โ€” and LATTICE++ (fused with cheap retrieval) reaches 49.1.** Budget behavior: "reranking offers a better tradeoff at low token budgets, but LATTICE converges to a higher asymptote after a moderate budget." + +This is the strongest single result in reasoning-aware retrieval, and it is under-cited: **an untrained LLM doing hierarchical search matches purpose-trained reasoning-retrieval ensembles.** It reframes the whole area as a test-time-compute allocation problem rather than a representation-learning problem. + +**Reranker-Guided Search (RGS)** โ€” Haike Xu, Tong Chen. arXiv:2509.07163, 8 Sep 2025. Greedy search on proximity graphs to select *which* documents to send to the reranker, rather than reranking a fixed top-k. **+3.5 BRIGHT, +2.9 FollowIR, +5.1 M-BEIR**, all within a 100-document reranker budget. + +**State Machine Reasoning (SMR)** โ€” Dohyeon Lee, Yeonseok Jeong, Seung-won Hwang. arXiv:2505.23059, 29 May 2025. Discrete Refine/Rerank/Stop actions with early stopping instead of free-form CoT. On BEIR and BRIGHT: **+3.4% nDCG@10 while cutting token usage 74.4%.** Generalizes across LLMs and retrievers without task-specific tuning. The cleanest "overthinking is real in IR" result. + +**Verbal-R3** โ€” Sangkwon Park, Donghun Kang, Jisoo Mok, Sungroh Yoon (SNU). arXiv:2605.01399, 2 May 2026, **ACL 2026 Main**. A "Verbal Reranker" emitting both relevance scores and analytic narratives connecting query to context, plus **relevance-guided test-time scaling** for trajectory expansion. + +**โญ Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction.** Zhuofeng Li, Haoxiang Zhang, Cong Wei, Pan Lu, Ping Nie, Yi Lu, Yuyang Bai, Shangbin Feng, Hangxiao Zhu, Ming Zhong, Yuyu Zhang, Jianwen Xie, **Yejin Choi, James Zou, Jiawei Han, Wenhu Chen, Jimmy Lin**, Dongfu Jiang, Yu Zhang. arXiv:2605.05242, 3 May 2026. Agents search raw corpora with **grep and shell commands** โ€” no embedding model, no vector index. Claims DCI "substantially outperforms sparse, dense, and reranking baselines" on BRIGHT and BEIR, and performs strongly on BrowseComp-Plus and multi-hop QA **without any semantic retriever**. Thesis: "retrieval quality depends not only on reasoning ability but also on the **resolution of the interface** through which models access corpora." โš ๏ธ **I could not extract the numeric tables** โ€” the arXiv HTML 404s and the PDF's tables are in compressed streams. Existence, authorship, date, and qualitative claims verified; **the numbers are not**. Given the author list this deserves a follow-up read. + +**Orion** โ€” Supriti Vijay, Aman Priyanshu, Anu Vellore, Baturay Saglam, Amin Karbasi. arXiv:2511.07581, 10 Nov 2025. 350Mโ€“1.2B models doing iterative retrieval via synthetic trajectories + SFT + RL + inference-time beam search: **SciFact 77.6 (vs 72.6), BRIGHT 25.2 (vs 22.1), NFCorpus 63.2 (vs 57.8)**, beating retrievers 200โ€“400ร— larger on 5 of 6 benchmarks with 3% of the training data. + +**โš ๏ธ Withdrawn paper:** "Adaptive Retrieval for Reasoning-Intensive Retrieval" (REPAIR), arXiv:2601.04618, submitted 8 Jan 2026, **withdrawn by the authors 14 Apr 2026**. It claimed +5.6pp. Do not cite it. + +**โ†’ OPEN DEBATE #2 โ€” is reasoning-aware retrieval a training problem or a test-time-compute problem?** ReasonIR/RaDeR/BGE-Reasoner say train the retriever. LATTICE (Google) matches the best trained ensembles with an **off-the-shelf** LLM. BrowseComp-Plus says the reasoning-specialized retriever (ReasonIR-8B) **loses to a general scaled embedder (Qwen3-8B)** inside an agentic loop. SMR says a large fraction of the reasoning tokens are pure waste (โˆ’74.4% tokens for +3.4% nDCG). The training camp has never been evaluated against the test-time-compute camp under a matched compute budget. + +--- + +## 3. RETRIEVAL ARCHITECTURE COMPONENTS + +### 3.1 Late-interaction revival + +**The modern models.** *GTE-ModernColBERT-v1* (LightOn, on `gte-modernbert-base`, 128-dim/token): **BEIR avg nDCG@10 = 54.67** vs answerai-colbert-small 53.79 reported / 53.35 on LightOn's rerun. LongEmbed(32k) mean 88.39. โš ๏ธ Vendor model-card numbers; the **+0.88 to +1.32 margin is under 1.5 points**, and the card's rerun of the competitor scores below the competitor's published figure โ€” the classic pattern warranting skepticism. + +**The best-documented efficiency table in the lineage** โ€” *mxbai-edge-colbert-v0* (Rikiya Takehi, Benjamin Claviรฉ, Sean Lee, Aamir Shakir; Mixedbread/Answer.AI). arXiv:2510.14880, 16 Oct 2025: + +| model | params | dim | BEIR avg | LongEmbed 32k | CPU time | Mem/10k docs | +|---|---|---|---|---|---|---| +| mxbai-edge-colbert-17m | 17M | 48 | 0.490 | 0.847 | 487s | **275 MB** | +| mxbai-edge-colbert-32m | 32M | 64 | 0.521 | 0.849 | 589s | 366 MB | +| ColBERTv2 | 130M | 128 | 0.488 | 0.428 | **1540s** | **732 MB** | +| answerai-colbert-small-v1 | 33M | 96 | 0.534 | โ€” | 621s | 549 MB | +| GTE-ModernColBERT-v1 | 130M+ | 128 | 0.547 | 0.898 | โ€” | โ€” | + +**The efficiency lineage, verified:** +- **XTR** โ€” Jinhyuk Lee, Zhuyun Dai, Sai Meher Karthik Duddu, Tao Lei, Iftekhar Naim, Ming-Wei Chang, Vincent Y. Zhao (Google DeepMind). arXiv:2304.01982, **NeurIPS 2023**. +2.8 BEIR nDCG@10, scoring **2โ€“3 orders of magnitude cheaper** than ColBERT. +- **MUVERA** โ€” Laxman Dhulipala, Majid Hadian, Rajesh Jayaram, Jason Lee, Vahab Mirrokni (Google Research). arXiv:2405.19504, 29 May 2024, **v2 8 Jun 2026 corrected the Theorem 2.1 dimension bound**. Reduces multi-vector to single-vector MIPS via asymmetric Fixed Dimensional Encodings. **2โ€“5ร— fewer candidates at equal recall; 10% higher recall with 90% lower latency** vs PLAID across BEIR. Now theoretically bracketed by the same group: arXiv:2607.20393 proves near-matching lower bounds (MUVERA is near-optimal for FDE-style reductions); arXiv:2606.23475 proves multi-vector is formally more expressive than single-vector. +- **CRISP** โ€” Veneroso, Jayaram, Rao, Hernรกndez รbrego, Hadian, Cer (Google Research). arXiv:2505.11471, 16 May 2025. Clustering trained *into* the model: **~3ร— vector reduction while beating the unpruned model**; 11ร— at 3.6% loss. +- **WARP** โ€” Jan Luca Scheerer, Matei Zaharia, Christopher Potts, Gustavo Alonso, **Omar Khattab**. arXiv:2501.17788, **SIGIR 2025**. **41ร— latency reduction vs XTR reference; 3ร— over ColBERTv2/PLAID.** +- **ColBERT-serve** (arXiv:2504.14903): memory-mapped scoring, **90% RAM reduction**. **Constant-space multi-vector** โ€” MacAvaney, Mallia, Tonellotto, **ECIR 2025** (arXiv:2504.01818): fixed vector count decoupled from doc length. โš ๏ธ exact deltas in PDF, unverified. +- 2026 kernels/indexes: **ColBERTSaR** (Eugene Yang, Andrew Yates, Dawn Lawrie, Mayfield, Samuel, Jha, JHU HLTCOE; arXiv:2606.05568) **50โ€“70% smaller than a 1-bit PLAID index**; **No More K-means** (arXiv:2605.30120, **ICML 2026**) **15ร— faster indexing than ColBERTv2**; **TileMaxSim** (arXiv:2606.26439) 100K-candidate scoring **268ms โ†’ 1.2ms** on H100; **FLASH-MAXSIM** (IBM, arXiv:2605.29517) **9ร— less inference / ~100ร— less training memory** at ColPali scale; **FastLane** (Ramnath Kumar, Prateek Jain, Cho-Jui Hsieh, Google; arXiv:2601.06389) **up to 30ร— lower compute**; **LEMUR** (arXiv:2601.21853) order-of-magnitude faster MV search; **PLAID-PRF** (Xiao Wang, MacAvaney, Macdonald; arXiv:2607.18626) **+4.3% nDCG@10** from pseudo-relevance feedback over PLAID centroids. + +**Honest 2026 storage/latency multiplier.** Naive ColBERTv2 at 128-dim fp16/token is **~732 MB per 10k docs** vs ~15 MB for a 768-dim single-vector index โ€” that is the folk "50โ€“100ร—". Stack the 2025โ€“26 techniques (dim 128โ†’48, small backbone, CRISP training-time clustering, PQ) and you land at **~3โ€“10ร—**. Anyone quoting 100ร— in 2026 is citing ColBERTv1. Latency: PLAID's fastest measured point is **73โ€“80.5 ms/query** on MS MARCO; WARP claims 3ร— on top; **~10โ€“30 ms/query** is the current well-engineered figure. + +**โญ The strongest paper arguing late interaction is NOT worth it.** "A Reproducibility Study of PLAID" โ€” **Sean MacAvaney & Nicola Tonellotto, SIGIR 2024 Reproducibility Track**, arXiv:2404.14989. + +| PLAID setting | nprobe | t_cs | ndocs | latency | DL19 nDCG@10 | +|---|---|---|---|---|---| +| (a) | 1 | 0.50 | 256 | 80.5 ms | 0.739 | +| (b) | 2 | 0.45 | 1024 | 103.4 ms | 0.745 | +| (c) | 4 | 0.40 | 4096 | 163.9 ms | 0.745 | + +**Re-ranking a BM25 candidate list with ColBERTv2 runs at as low as 9 ms/query at n=200, vs 73 ms/query for the fastest PLAID pipeline โ€” an ~8ร— latency advantage โ€” reaching RR@10 = 0.373 on MS MARCO Dev**, which the authors note beats early BERT cross-encoders. PLAID only wins at high latency where lexical recall binds. Their mechanistic finding: **most PLAID token clusters are predominantly aligned with a single token** โ€” the centroid machinery approximates lexical matching. Their charge is methodological: prior late-interaction efficiency work **omitted the obvious baseline**. **This has not been rebutted with an updated head-to-head.** + +Corroborating skepticism: *Are LLM-Based Retrievers Worth Their Cost?* (Abdallah, Holdcroft, Ali, Jatowt, **SIGIR 2026**, arXiv:2604.03676) โ€” 14 retrievers ร— 12 BRIGHT tasks; large LLM bi-encoders incur substantial latency for modest gains, reasoning augmentation shows diminishing returns, and **confidence calibration is weak across all families**, so raw scores are unreliable for routing. *KaLM-Reranker-V1: **Fast but Not Late Interaction*** (arXiv:2606.22807). *MICE* (arXiv:2602.16299): **4ร— lower latency than standard cross-encoders while matching ColBERT-class quality**. *MINER* (arXiv:2605.06460): narrows the MV-to-dense gap to **0.2 nDCG@5**. *Your Embedding Model is SMARTer Than You Think* (arXiv:2605.24938): late interaction over **frozen hidden states** of existing single-vector models, no dedicated MV index needed. + +And the proponents concede it: the **LIR @ ECIR 2026** workshop proposal (Benjamin Claviรฉ, Xianming Li, Antoine Chaffin, **Omar Khattab**, Tom Aarsen, Manuel Faysse, Jing Li; arXiv:2511.00444) says these models pose "significant challenges of efficiency, usability, and integration" and "prohibitive storage and computational overhead," and explicitly solicits **negative or puzzling results**. + +**โญ The strongest pro-multi-vector argument is theoretical.** "On the Theoretical Limitations of Embedding-Based Retrieval" โ€” **Orion Weller, Michael Boratko, Iftekhar Naim, Jinhyuk Lee (Google DeepMind + JHU)**, arXiv:2508.21038, v2 12 Mar 2026, **ICLR 2026**. The number of top-k document subsets returnable by *any* query is bounded by embedding dimension. LIMIT (50k docs, 1000 queries, k=2): + +| model | R@2 | R@10 | R@100 | +|---|---|---|---| +| **BM25** | **97.8** | **100.0** | **100.0** | +| GTE-ModernColBERT (MV) | 23.1 | 34.6 | **54.8** | +| Promptriever Llama3 8B | 3.0 | 6.8 | 18.9 | +| GritLM 7B | 2.4 | 4.1 | 12.9 | +| Gemini Embedding | 1.6 | 3.5 | 10.0 | +| E5-Mistral 7B | 1.3 | 2.2 | 8.3 | +| Qwen3 Embedding | 0.8 | 1.8 | 4.8 | + +โš ๏ธ **LIMIT is adversarially constructed to hit the bound.** It proves an existence claim, not a claim about natural query distributions. And it is contested: **Bangachev, Bresler, Kogan, Polyanskiy (MIT)**, arXiv:2605.23556, 22 May 2026, don't dispute the bound but prove near-optimal margins are achievable at **d = O(k log(n/k))** in the sparse regime (necessary and sufficient), margin ฮ˜(k^(-1/2)), holding to trillions of points โ€” i.e. dโ‰ˆ1000 is provably near-sufficient. Separately, *Spectral Retrieval* (arXiv:2605.24764) lifts LIMIT-small R@10 from **0.33 to 0.90 without retraining**, suggesting part of the failure is a scoring-function artifact. + +**โ†’ OPEN DEBATE #3.** Weller et al. (DeepMind, ICLR'26): single-vector is dimensionally capped, MV is the escape. Bangachev et al. (MIT): the cap is far looser than worst-case. MacAvaney & Tonellotto (SIGIR'24): even granting MV's quality edge, **BM25 + ColBERT reranking dominates the Pareto frontier at deployable latencies**. No 2026 paper reconciles the three. + +### 3.2 Hybrid sparse+dense fusion + +**The canonical fusion-function result is still pre-2025 and still unrebutted.** "An Analysis of Fusion Functions for Hybrid Retrieval" โ€” Sebastian Bruch, Siyu Gai, Amir Ingber, **ACM TOIS Aug 2023**, arXiv:2210.11934. RRF is **sensitive to its parameters**; convex combination is **agnostic to score-normalization choice** (min-max, z-score, any linear transform are rank-equivalent); **CC beats RRF in- and out-of-domain** and is sample-efficient. โš ๏ธ **Practice/theory gap:** essentially every 2025โ€“26 applied paper uses RRF anyway, because it needs no tuning data. Nobody has re-tested. + +**2025โ€“26 evidence:** + +| Paper | Date | Finding | +|---|---|---| +| From Retrieval to Generation (Abdallah, Mozafari, Piryani, Ali, Jatowt), arXiv:2502.20245 | Feb 2025 | **BEIR nDCG@10: BM25 43.42 โ†’ hybrid 52.59 (+9.17)** | +| From BM25 to Corrective RAG: Text-and-Table (Akarsu, Karaman, Mierbach), arXiv:2604.01733 | Apr 2026 | **23,088 financial QA queries** โ€” largest here. Two-stage hybrid + neural rerank Recall@5 = **0.816**. **BM25 outperforms dense on financial documents.** Recommends hybrid RRF + cross-encoder as the minimum viable baseline | +| Dissecting Agentic RAG, arXiv:2606.21553 | Jun 2026 | **Fixed hybrid RRF beats rule-based adaptive routing (+1.8 EM, +1.9 F1)**; EM 53.2% HotpotQA. Negative result for adaptive routing | +| KohakuRAG (Yeh, Ku, Huang, Tu), arXiv:2603.07612 | Mar 2026 | โš ๏ธ **Contrarian: hierarchical dense alone matches hybrid; BM25 adds only +3.1pp** | +| Training-Free Lexical-Dense Fusion for Conversational Memory, arXiv:2606.04194 | Jun 2026 | Late-interaction dense + BM25: **+8.8 to +17.2 Hit@1**, no training | +| DAT: Dynamic Alpha Tuning (Hsu, Tzeng), arXiv:2503.23013 | Mar 2025 | LLM-judged per-query dense/BM25 weighting beats fixed-weight hybrid | +| Hybrid Retrieval for Hallucination Mitigation (ISTI-CNR), arXiv:2504.05324 | 2025 | On HaluBench, hybrid gives highest accuracy on fails and lowest hallucination + rejection rates | + +**Where BM25 still wins outright:** LIMIT (97.8 R@2 vs 0.8 for the best dense); financial text-and-table at 23k queries; low-latency first-stage (9 ms/q + rerank beating the entire PLAID frontier below ~70 ms/q). **Where it clearly loses:** aggregate suites โ€” HAKARI-Bench (551 tasks, 43 languages) puts BM25 at **50.24** macro nDCG@10ร—100 vs best sub-1B dense **64.93**; and agentic loops โ€” BrowseComp-Plus, GPT-5 at **55.9% with BM25 vs 70.12% with Qwen3-Embedding-8B**, with the authors noting BM25's documents are "less useful in the iterative deep research process." + +**โ†’ OPEN DEBATE #4.** BM25's standing is strongly task-conditional; blanket claims fail in both directions. And **+9.17 (arXiv:2502.20245) vs +3.1pp (KohakuRAG) for the hybrid gain is a direct, unexplained, unreplicated conflict** โ€” plausibly a corpus-structure effect where hierarchical indexing already captures what BM25 contributes. + +**Reproducibility infrastructure:** Pyserini (Lin et al., SIGIR 2021) remains the reference. โš ๏ธ "Gosling Grows Up: Retrieval with Learned Dense and Sparse Representations Using Anserini" (Lin group, SIGIR 2025) surfaced in search only โ€” arXiv ID unverified, not guessed. *GPUSparse* (arXiv:2606.26441): **235ร— speedup over Pyserini CPU at 8.8M docs** โš ๏ธ single-author, unreviewed. + +### 3.3 Learned sparse + +**SPLADE-v3** โ€” Carlos Lassance, Hervรฉ Dรฉjean, Thibault Formal, Stรฉphane Clinchant (Naver Labs Europe). arXiv:2403.06789, 11 Mar 2024. **>40 MRR@10 on MS MARCO dev; +2% out-of-domain on BEIR** over SPLADE++. Meta-analysis over **40+ query sets**: statistically significantly better than BM25 and SPLADE++, **competitive with cross-encoder rerankers**. + +**โญ The 2026 headline: LACONIC** โ€” Zhichao Xu, Shengyao Zhuang, Crystina Zhang, Xueguang Ma, Yijun Tian, Maitrey Mehta, **Jimmy Lin**, Vivek Srikumar (Utah + CSIRO + Waterloo). arXiv:2601.01684, 4 Jan 2026. Two-phase curriculum on Llama-3 1B/3B/8B: weakly-supervised pre-finetuning for bidirectional contextualization, then hard-negative finetuning. **8B: 60.2 nDCG on MTEB Retrieval, ranked 15th as of 1 Jan 2026, with 71% less index memory than an equivalent dense model**, running on standard CPU hardware. โš ๏ธ Self-reported rank; "15th" also means 14 dense models beat it. + +**SPLARE** โ€” Thibault Formal, Antoine Louis, Hervรฉ Dรฉjean, Stรฉphane Clinchant (Naver Labs). arXiv:2603.13277, **ICLR 2026**. Replaces the vocabulary projection with **sparse-autoencoder features**; SPLARE-7B posts top results on MMTEB multilingual + English retrieval. + +Other verified: **CSPLADE** (Zhichao Xu, Aosong Feng, Yijun Tian, Haibo Ding, Lin Lee Cheong, Amazon; arXiv:2504.10816, **IJCNLP-AACL 2025 Main**) โ€” 8B-scale LSR, fixes early-stage contrastive instability and unidirectional attention. **Li-LSR** (arXiv:2505.01452) โ€” **inference-free query encoding** via table lookup, **+1โ€“1.8 nDCG over Splade-v3-Doc**. **Sparton** (arXiv:2603.25011) โ€” fused Triton kernel, **+33% batch size, 14% faster training**. **V-SPLADE** (Naver, arXiv:2605.30917) โ€” inference-free multimodal LSR for production visual document search. **MILCO** (arXiv:2510.00671) โ€” multilingual LSR via a shared English lexical space. **UEmbed** (Alibaba/Tongyi, Pengjun Xie et al., arXiv:2608.02583) โ€” decoder-only model emitting sparse and dense simultaneously, 2Bโ€“9B. โš ๏ธ Skeptical note: *Understanding Wacky Weights* (Polyakov, Scells, Eickhoff, arXiv:2605.19628) finds larger vocabularies correlate with **semantically unrelated** expansion terms โ€” SPLADE is less interpretable than assumed. + +**Efficiency asymmetry worth knowing** (from HAKARI-Bench on SPLADE-v3): **document-side pruning costs only +0.01โ€“0.04 points** at d=256โ†’512, while **query-side reduction costs +2.5โ€“3.6 points** at q=8โ†’32. + +**โ†’ EMERGING SYNTHESIS.** ColBERTSaR (JHU, arXiv:2606.05568) demonstrates **ColBERT with product quantization is equivalent to learned-sparse retrieval**. Combined with MacAvaney & Tonellotto's finding that PLAID clusters align ~1:1 with tokens, there is a coherent thread arguing **late interaction and learned sparse are converging on the same mechanism** โ€” which would dissolve Debate #3 into an implementation question. Not yet consensus. + +### 3.4 Embedding model scaling and MTEB standings + +โš ๏ธ **I could not scrape the live MTEB leaderboard.** The HF Space is a client-rendered Gradio app; every fetch returned the loading shell. **I will not state a mid-2026 #1 as fact.** + +**Verified anchors:** *Qwen3-Embedding-8B* โ€” **70.58 MTEB Multilingual, No.1 as of 5 Jun 2025** (arXiv:2506.05176, Alibaba Tongyi; 0.6B/4B/8B, 119 languages). *Gemini Embedding* โ€” **68.32 Task Mean on MTEB(Multilingual), highest at time of writing** (Jinhyuk Lee, Feiyang Chen, Sahil Dua, Daniel Cer + 43 co-authors, Google DeepMind; arXiv:2503.07891, 10 Mar 2025; size undisclosed). Both self-reported and now stale. + +**Verified mid-2026 cross-vendor table** (jina-embeddings-v5-text, Jina AI, arXiv:2602.15547, 17 Feb 2026): + +| model | params | MMTEB avg | MTEB-Eng | Retrieval nDCG@10 | +|---|---|---|---|---| +| Qwen3-4B (teacher) | 4B | **69.5** | **74.6** | **69.60** | +| jina-v5-text-small | 677M | 67.0 | 71.7 | 64.88 | +| jina-v5-text-nano | 239M | 65.5 | 71.0 | 63.26 | +| Qwen3-0.6B (instruct) | 596M | 64.3 | 70.5 | 64.65 | +| multilingual-e5-large-instruct | 560M | 63.2 | 65.5 | 57.12 | +| Qwen3-0.6B (generic) | 596M | 61.1 | 67.0 | โ€” | +| embeddinggemma-300m | 308M | 61.1 | 69.7 | 62.49 | +| voyage-4-nano | 340M | 58.9 | 63.3 | 63.58 | +| jina-v3 | 572M | 58.4 | 65.7 | 55.76 | +| snowflake-arctic-embed-l-v2 | 568M | 57.0 | 63.6 | 58.36 | + +โš ๏ธ Vendor-authored; comparison set deliberately sub-1B except the teacher. + +**Independent 2026 comparison: HAKARI-Bench** (Yuichi Tateno, arXiv:2606.22778, 22 Jun 2026) โ€” 35 benchmarks / 551 tasks / 43 languages / 55 models, nano-sets validated at **Spearman 0.983 vs MTEB v2, 0.975 vs MMTEB v2, 0.973 vs full BEIR**. Top by macro nDCG@10ร—100: **jina-v5-text-small 64.93**, jina-v5-text-nano 63.80, **microsoft/harrier-oss-v1-0.6b 63.68**, **perplexity-ai/pplx-embed-v1-0.6b 63.64**, embeddinggemma-300m 62.58 โ€” *BM25 baseline 50.24*. Rerankers beat all: Qwen3-Reranker-0.6B **68.03**. Quantization deltas (mean over 33 models): binary **โˆ’6.50**, int8 **โˆ’1.95**, binary+rescore **โˆ’0.93**, **int8+rescore โˆ’0.09 (effectively lossless)**. โš ๏ธ Single-author, โ‰ค~1B scope, nano-sets not full benchmarks โ€” but the only independent unified-conditions 2026 comparison found. + +**Net honest answer:** among โ‰ค1B open-weight models, **jina-embeddings-v5-text-small** sits at or near the top on two independent 2026 sources. Overall, **Qwen3-Embedding-8B / Qwen3-4B** remain the reference ceiling with **Gemini Embedding** the strongest closed model on last-verified data. + +**โญ Scaling laws.** "Scaling Laws for Embedding Dimension in Information Retrieval" โ€” Julian Killingback, Mahta Rafiee, Madine Manas, **Hamed Zamani** (UMass CIIR). arXiv:2602.05062, 4 Feb 2026. Retrieval performance vs embedding dimension **fits a power law**, with predictive models on dimension alone and jointly with model size. **Aligned tasks: monotone improvement with diminishing returns. Misaligned tasks: unpredictable โ€” larger embedding dimension can actively degrade results.** That second half is the non-obvious finding and directly cautions against "bigger dim is safer." Adjacent: *Retrieval Capabilities of LLMs Scale with Pretraining FLOPs* (arXiv:2508.17400); *BitNet Text Embeddings* (Zhen Li, Xin Huang, Liang Wang, Nan Yang, **Furu Wei**, MSR; arXiv:2606.25674) โ€” extreme quantization "largely comparable" to full-precision teachers on MMTEB(eng,v2). Low end: *Bekko Embedding* (arXiv:2607.25180) โ€” **8M active params โ†’ 56.2 MMTEB Multilingual v2; 25M โ†’ 57.5** (vs BM25's 50.24). + +**โญ The leaderboard-integrity failure, primary-source verified.** GitHub `embeddings-benchmark/mteb` **issue #3934**, "Decision: Temporary removal of the private RTEB column", **14 Jan 2026**. RTEB was built with private test sets specifically to defeat overfitting โ€” but it was **co-developed with Voyage AI (since acquired by MongoDB), who therefore had direct access to the private evaluation data while competing on the leaderboard.** The MTEB team's words: *"the uneven playing field fundamentally undermines trust in MTEB leaderboards, which is unacceptable for a community benchmark."* No misuse alleged; the structural conflict alone forced the call. The private column was removed; it returns once the private pool is diversified with contributions from orgs that don't ship competing models. + +**Peer-reviewed critiques, six independent groups reaching the same conclusion:** *On the Robustness of Multilingual Text Embedding Rankings* (Gjorgjevikj, Korouลกiฤ‡ Seljak, Eftimov, arXiv:2605.31142) โ€” **MTEB rankings are sensitive to dataset composition and aggregation method; conclusions lack robustness**. *MTEB-BR* (arXiv:2607.04581) โ€” multilingual leaderboard correlates only **ฯ=0.75** with Brazilian Portuguese performance. *MTEB-PT* (arXiv:2607.04071) โ€” same for Portuguese. *LMEB* (Xinping Zhao et al., arXiv:2603.12572) โ€” long-horizon memory results **orthogonal to MTEB; larger models don't reliably win**. *STEB* (Rivera Soto, Wegmann, Aggazzotti, arXiv:2606.31741) โ€” semantic embeddings **consistently fail on stylistic tasks**. *PosIR* (arXiv:2601.08363, 310 datasets, 10 MMTEB SOTA models) โ€” **position bias is pervasive and invisible to MTEB**. *CS-MTEB* (arXiv:2604.17632) โ€” **up to 27% degradation** on code-switched queries. *SABER-Math* (arXiv:2606.29894) โ€” "general-purpose IR benchmarks such as MTEB do not reliably predict mathematical performance." Also *HTEB* (arXiv:2605.28190) and *PTEB* (arXiv:2510.06730) proposing stochastic re-paraphrasing at eval time. + +**The best-supported claim in this whole section: a high MTEB/MMTEB average does not transfer to your specific language, domain, or task.** + +โš ๏ธ **Instruction-following embeddings โ€” partial coverage.** Search budget ran out before a dedicated pass. What is verified: instruction conditioning is worth **+3.2 MMTEB / +3.5 MTEB-Eng at identical parameter count** (Qwen3-0.6B instruct 64.3/70.5 vs generic 61.1/67.0); **Promptriever Llama3 8B was the best single-vector model on LIMIT (R@100 = 18.9 vs Qwen3's 4.8)**, i.e. instruction-following retrievers are meaningfully more robust to the dimensional bound; and **MMTEB** (Kenneth Enevoldsen + 64 co-authors, **ICLR 2025**, arXiv:2502.13595, 500+ tasks / 250+ languages) added instruction-following as a task category and found **smaller multilingual models often outperform large LLMs**. + +--- + +## 4. CONTEXT HANDLING + +### 4.1 Late chunking + +**The primary paper is arXiv-only, from the vendor that sells the embeddings it was tested on.** "Late Chunking: Contextual Chunk Embeddings Using Long-Context Embedding Models" โ€” Michael Gรผnther, Isabelle Mohr, Daniel James Williams, Bo Wang, Han Xiao (Jina AI). arXiv:2409.04701, v1 7 Sep 2024, v3 7 Jul 2025. Comments field says "11 pages, 3rd draft." **No venue.** It has an OpenReview page (74QmBTV0Zf) with no acceptance record. + +Verified numbers (Table 2, nDCG@10, fixed 256-token boundaries, naive โ†’ late): + +| Dataset | jina-v2-small | jina-v3 | nomic-v1 | +|---|---|---|---| +| SciFact | 64.2 โ†’ 66.1 | 71.8 โ†’ 73.2 | 70.7 โ†’ **70.6 (โ†“)** | +| NFCorpus | 23.5 โ†’ 30.0 | 35.6 โ†’ 36.7 | 35.3 โ†’ **35.3 (=)** | +| FiQA | 33.3 โ†’ 33.8 | 46.3 โ†’ 47.6 | 37.0 โ†’ 38.3 | +| TRECCOVID | 63.4 โ†’ 64.7 | 73.0 โ†’ 77.2 | 72.9 โ†’ 75.0 | +| **Average** | โ€” | โ€” | **52.2 โ†’ 54.0** | + +**The headline effect is +1.5 to +1.9 nDCG@10 absolute (2.7โ€“3.6% relative)**, and it is not uniform. The authors' own conceded failure case: late chunking **hurts on synthetic needle tasks** (Needle-8192, Passkey-8192), because contextualizing a planted needle with unrelated filler dilutes it. Their comparison against contextual retrieval (Table 4) is **a single anecdotal example with one query** โ€” cosine 0.6343 naive โ†’ 0.8516 late โ†’ 0.8590 contextual. Not an experiment. Their argument against contextual retrieval is cost, not accuracy. + +**โญ The independent replication is mixed-to-negative.** "Reconstructing Context: Evaluating Advanced Chunking Strategies for RAG" โ€” Carlo Merola, Jaspinder Singh. arXiv:2504.19754, **2nd Workshop on Knowledge-Enhanced IR, ECIR 2025**. NDCG@5, early โ†’ late: NFCorpus Stella-V5 0.443โ†’0.445; jina-v3 0.374โ†’0.380; jina-v2 0.261โ†’0.280; **BGE-M3 0.246 โ†’ 0.070**. MSMarco, Stella-V5: **0.630 โ†’ 0.503**. + +**This is the single most important refutation datapoint: late chunking is model-dependent, can catastrophically fail (BGE-M3, ~72% relative collapse), and loses badly on MSMarco with a strong non-Jina encoder.** The gains appear to hold mainly on small domain-specific corpora with Jina's own models. โš ๏ธ Workshop paper, small scale; the BGE-M3 collapse is plausibly a pooling-compatibility artifact. But nobody has published a rebuttal, which is itself informative. Same paper's head-to-head: contextual retrieval NDCG@5 **0.317** vs late chunking **0.309** on an NFCorpus subset โ€” a ~2.6% relative edge at much higher cost. + +**โญ The best systematic 2026 evaluation.** "Beyond Chunk-Then-Embed: A Comprehensive Taxonomy and Evaluation of Document Chunking Strategies for IR" โ€” Yongjie Zhou, Shuai Wang, Bevan Koopman, **Guido Zuccon** (Queensland/CSIRO). arXiv:2602.16974, 19 Feb 2026. Abstract, verbatim: *"Contextualized chunking improves in-corpus effectiveness but degrades in-document retrieval."* + +- **In-corpus (BEIR, nDCG@10):** jina-v3 โ€” **Paragraph 0.4948** > Fixed 0.4849 > Semantic 0.4726 โ‰ˆ Sentence 0.4723 > LumberChunker 0.4690 > Proposition 0.3888. **Structure-based beats LLM-guided by 5โ€“27%.** +- **In-document (GutenQA, 3k QA over 100 books, DCG@10):** LumberChunker wins decisively โ€” jina-v3 0.5640 vs Paragraph 0.4574. +- **Effect of late chunking:** in-corpus, Proposition **+22.87% to +26.94%**; LumberChunker +2.42% to +4.80%; structure-based **0% to +7.20%**. In-document, Paragraph **โˆ’10.76% to โˆ’62.47%**; Sentence โˆ’5.39% to โˆ’62.57%. +- Throughput: LumberChunker **1.11 docs/s vs paragraph-based 1,854 docs/s โ€” ~1,600ร— slower.** + +Late chunking's benefit is largest exactly where naive chunks are most context-starved, near-zero where chunks are already coherent, and **actively harmful for needle-style in-document retrieval** โ€” independently corroborating the authors' own caveat. + +**The chunking-strategy benchmark literature.** *Is Semantic Chunking Worth the Computational Cost?* (Renyi Qu, Ruixuan Tu, Forrest Sheng Bao; arXiv:2410.13070, **Findings of NAACL 2025**) โ€” cost not justified; fixed ~200-word chunks match or beat it. *Chunking Methods on RAG: Effectiveness vs Computational Cost* (Wrocล‚aw UST, arXiv:2606.00881) โ€” fixed-size and recursive-semantic most stable; **LumberChunker highest answer quality but completed on only ~30% of datasets** (timeouts); runtimes <1s vs **8.37 h**; DenseX 15+ h; only 5 of 8 methods completed everywhere. *A Systematic Investigation of Document Chunking Strategies* (Shaukat, Adnan, Kuhn, U. Canberra; arXiv:2603.06976) โ€” largest sweep: 36 approaches ร— 6 domains ร— 5 embedders = 1,080 configs; **Paragraph Group Chunking best (mean nDCG@5 โ‰ˆ 0.459, P@1 24%), naive fixed-*character* worst (nDCG@5 < 0.244, P@1 โ‰ˆ 2โ€“3%)**. *Evaluating Chunking Strategies for RAG on Academic Texts* (arXiv:2607.01852) โ€” cluster-based semantic chunking **did not outperform** fixed-size or recursive; also warns **RAGAs faithfulness "shows limited reliability in this setup."** + +**โ†’ OPEN DEBATE #5, and it dissolves on inspection.** Four papers say simple wins; Shaukat et al. reports a ~2ร— nDCG spread implying chunking is a vital lever. The reconciliation: **Shaukat's worst baseline is fixed-*character* splitting (which shreds words), while the "simple wins" camp baselines against fixed-token or recursive splitting.** Consensus as of mid-2026: **paragraph- or sentence-respecting structural chunking is a strong, near-optimal, essentially free baseline; embedding-similarity semantic chunking and LLM-guided chunking do not reliably beat it in-corpus at 100โ€“1,600ร— the cost.** + +**Descendants.** *Context is Gold to find the Gold Passage* (Conti, Faysse, Viaud, Bosselut, Hudelot, Colombo; arXiv:2505.24782) โ€” introduces **ConTEB** benchmark and **InSeNT** in-sequence-negative contrastive post-training combined with late-chunking pooling; the most credible academic endorsement of the pooling operator, **but it requires training โ€” zero-shot late chunking is the weaker claim.** Also *ColChunk* visual late chunking (arXiv:2604.10167), *Graph-Aware Late Chunking for Biomedical* (arXiv:2603.22633), *pplx-embed* (Perplexity, arXiv:2602.11151 โ€” uses late chunking in production; โš ๏ธ Bo Wang is a Jina co-author, not independent). + +### 4.2 Contextual retrieval + +**The source is a blog post. There is no paper.** Anthropic engineering blog, 19 Sep 2024. Verified claims: top-20 retrieval **failure rate 5.7% โ†’ 3.7% (โˆ’35%)** with contextual embeddings; **โ†’ 2.9% (โˆ’49%)** adding contextual BM25; **โ†’ 1.9% (โˆ’67%)** adding reranking. Cost **$1.02 per million document tokens**. Anthropic's own caveat: performance varies by embedding/source combination, run your own evals. + +**No arXiv version, no peer review, no dataset release, and no independent reproduction of 35/49/67 exists as of August 2026.** Anyone quoting "67%" is quoting a vendor blog with an unreleased internal eval. + +What academic work reports: the only head-to-head (Merola & Singh, ECIR 2025 workshop) puts contextual retrieval **~2.6% relative ahead of late chunking on an NFCorpus subset**, at substantially higher cost. Zhou et al. (2026) find **structure-based segmentation beats LLM-guided methods by 5โ€“27% nDCG@10 in-corpus** at 1,600ร— the throughput; contextualization helps most for propositions (+23โ€“27%) and barely at all for paragraphs (0 to +7.2%). โš ๏ธ A single-author, non-peer-reviewed Spanish-language preprint (*More Context Is Not Better: The Vector Dilution Paradox*, arXiv:2601.08851) reports an inverted-U: **moderate injection +18% recall; past a Contextualization Injection Ratio > 0.4, precision drops 22%** on targeted queries. Treat the numbers as unverified, but the shape is consistent with Zhou et al. + +Cost reality the blog omits: contextual retrieval **doubles index-build storage and forces full re-contextualization whenever chunk boundaries change**. + +**โ†’ OPEN DEBATE #6.** Position A (Anthropic, unreplicated): 49โ€“67% failure reduction, cheap. Position B (Zhou et al. 2026): LLM-side chunk processing loses to paragraph splitting in-corpus by 5โ€“27%. Position C (Merola & Singh): best of the advanced methods, by ~1โ€“3% on a small subset. Position D (unreviewed): over-injection costs 22% precision. **Practitioner verdict: directionally sound, most valuable for short/proposition-sized chunks in anaphora-heavy corpora (filings, codebases). The 35/49/67 figures should be cited as vendor-internal, not established.** + +### 4.3 Context compression + +**The lineage, all verified:** + +| Method | Authors / lab | arXiv | Venue | Headline | +|---|---|---|---|---| +| LLMLingua | Jiang, Wu, Lin, Yang, Qiu (MSR) | 2310.05736 | **EMNLP 2023** | up to 20ร— compression, little loss | +| LongLLMLingua | Jiang, Wu, Luo, Li, Lin, Yang, Qiu (MSR) | 2310.06839 | **ACL 2024** | NaturalQuestions **+21.4% with ~4ร— fewer tokens**; **94.0% cost reduction** on LooGLE; **1.4โ€“2.6ร— latency** speedup | +| LLMLingua-2 | Pan, Wu, Jiang, Xia, Luo, Zhang, Lin, Rรผhle, Yang, C.-Y. Lin, Zhao, Qiu, Zhang (MSR + Tsinghua) | 2403.12968 | **Findings ACL 2024** | **3โ€“6ร— faster than prior compressors; 1.6โ€“2.9ร— end-to-end latency reduction** at 2โ€“5ร— | +| RECOMP | Fangyuan Xu, Weijia Shi, Eunsol Choi | 2310.04408 | ICLR 2024 | compression to **6%** with minimal loss; can emit empty string | +| xRAG | Cheng, Wang, Zhang, Ge, Chen, Wei, Zhang, Zhao (MSRA + PKU) | 2405.13792 | **NeurIPS 2024** | context โ†’ **one token**; **>10% avg improvement** on six tasks; **3.53ร— FLOPs reduction** | +| CompAct | Yoon, Lee, Hwang, Jeong, Kang | 2407.09014 | **EMNLP 2024** | **47ร— compression**, strongest on multi-hop | +| PISCO | Naver Labs Europe | 2501.16075 | **Findings ACL 2025** | **16ร— compression, 0โ€“3% loss**; +8% over prior compressors; 48 h on one A100 | + +**โญ Provence is the compression result that survives scrutiny.** Nadezhda Chirkova, Thibault Formal, Vassilina Nikoulina, Stรฉphane Clinchant (NAVER LABS Europe). arXiv:2501.16214, 27 Jan 2025, **ICLR 2025** (confirmed: poster 29557, OpenReview TDy5Ih78b4). Sentence-level pruning as binary sequence labeling on a DeBERTa cross-encoder, **fused with the reranker** so pruning is free in a pipeline that already reranks. One hyperparameter (threshold โˆˆ {0.1, 0.5}) that transfers across domains. + +LLM-Eval scores: **NQ 72.4 at 62.2% compression vs 71.8 full context** (pruning *improves* โ€” denoising); **HotpotQA 56.7 @ 66.4% vs 57.0 full** (โˆ’0.3); **PopQA 59.3 @ 68.6% vs 57.8 full** (+1.5). Baselines on NQ: LLMLingua-2 **59.5** @74%; LongLLMLingua **61.3** @69%; RECOMP-extractive **70.6** @44%; DSLR **71.7** @45%. **Cross-domain across 7 datasets โ€” NQ, HotpotQA, TyDi QA, PopQA, BioASQ (biomedical), SyllabusQA (education), RGB (news) โ€” "negligible to no drop."** Overhead essentially zero when unified with the reranker; generation speedups 1.2โ€“1.4ร— at batch 1, 1.9โ€“2.0ร— at batch 256. + +โš ๏ธ Caveat: 50โ€“80% compression is far less aggressive than CompAct's 47ร— or xRAG's one token. The OOD robustness comes partly from not compressing very hard. + +**โญ The systematic evaluation bug in the entire compression literature.** "Fixed RAG Compression Collapses Measured Reader Scaling" โ€” Sugam Panthi, Rabab Abdelfattah. arXiv:2606.21807, 20 Jun 2026. A *fixed* compressor helps weak readers (removes noise they can't filter) and hurts strong readers (removes detail they could have used). Across **20 readers ร— 10 domain-method settings, compression gains decreased with reader baseline in 9 of 10 settings (p < 0.05)**. Generic summarization **flipped 31% of pairwise model rankings on LongMemEval-S**. A fixed HotpotQA compressor **obscured 80% of the Qwen-7B โ†’ GPT-4.1-mini improvement**. Pattern holds across compressor types and an external audit of **nine published papers**. Released `ragscale` (177k row-level transitions). + +**Every "Xร— compression with negligible loss" number above is reader-dependent, and the strong-reader case is systematically under-reported.** + +Also: *No Mean Feat: Simple, Strong Baselines for Context Compression* โ€” Yair Feldman, **Yoav Artzi** (Cornell Tech). arXiv:2510.20797, rev 10 May 2026. Introduces **BenchPress** and shows **mean pooling and a bidirectional compression-token variant strongly outperform the widely-used causal compression-token approach** โ€” the design underlying much of the gist-token line โ€” across scales, datasets, and ratios. *RAISE* (arXiv:2605.30029): 13 RAG algorithms ร— 7 datasets, "optimization performance is highly task-dependent" with **poor cross-dataset generalization**. *Control Under Compression* (arXiv:2608.01056): reliability "diverges sharply" in the 50โ€“35% retained-context band and compressor rankings are **not universal**. + +**2026 successors:** *CORE-RAG* (Cui, Weng, Tang, Liu, Li, He, Chen, Zhang, He, Ma; arXiv:2508.19282, v4 28 May 2026, **ICML 2026**) โ€” performance-driven compression, **at a 3% compression ratio, +3.3 EM over feeding full documents**. *ARC-Encoder* (Kyutai, arXiv:2510.20535) โ€” 4โ€“8ร— compression adapting to multiple decoders. *Sentinel* (arXiv:2505.23277) โ€” a **0.5B proxy achieves 5ร— compression competitive with 7B-scale methods**. Plus ECoRAG, ACC-RAG, AttnComp, EXIT, BRIEF (+3.0 EM / +4.16 F1 on HotpotQA), and *A Unified Model and Document Representation for On-Device RAG* (Killingback, Meshi, Li, **Zamani**, Karimzadehgan; arXiv:2604.14403 โ€” matches traditional RAG with **1/10 the context**). + +**โ†’ OPEN DEBATE #7.** The field publishes 16โ€“47ร— compression with 0โ€“3% loss and even "compression improves accuracy." Panthi & Abdelfattah show this is largely an artifact of weak readers. Feldman & Artzi show the dominant architectural choice is beaten by mean pooling. **Reading: extractive, query-conditioned, moderate compression (Provence-style, 50โ€“70%) is genuinely robust; soft/gist/extreme compression (xRAG one-token, CompAct 47ร—) has not been shown to survive either a strong reader or a domain shift.** + +### 4.4 Long context vs RAG โ€” what the evidence actually says + +**The 2024 axis of debate.** *Retrieval Augmented Generation or Long-Context LLMs?* โ€” Zhuowan Li, Cheng Li, Mingyang Zhang, Qiaozhu Mei, Michael Bendersky (Google DeepMind + Michigan). arXiv:2407.16833, **EMNLP 2024 industry track**. When sufficiently resourced, **LC consistently beats RAG on average** โ€” but **LC and RAG predictions are identical for >60% of queries**, and SELF-ROUTE achieves LC-comparable quality at **65% cost reduction (Gemini-1.5-Pro), 39% (GPT-4o)**. + +*In Defense of RAG in the Era of Long-Context LLMs* โ€” Tan Yu, Anbang Xu, Rama Akkiraju (NVIDIA). arXiv:2409.01666. **OP-RAG** keeps retrieved chunks in original document order rather than relevance order; reports an inverted-U in chunk count. โˆžBench EN.QA F1: Llama3.1-70B full context **34.26 @117K tokens**; GPT-4o 32.36 @117K; Gemini-1.5-Pro 43.08 @196K; SELF-ROUTE GPT-4o 34.95 @85K. **OP-RAG on Llama3.1-70B: 44.43 @16K, 45.45 @24K, 47.25 @48K.** EN.MC accuracy: full-context Llama3.1-70B 71.62 @117K vs **OP-RAG 88.65 @24K**. So **47.25 F1 with 48K tokens vs 34.26 F1 with 117K**. โš ๏ธ Single benchmark (book-length novel QA โ€” the format most favorable to retrieval), single model family, arXiv-only. + +**Databricks.** *Long Context RAG Performance of LLMs* โ€” Quinn Leng, Jacob Portes, Sam Havens, **Matei Zaharia**, Michael Carbin (Databricks Mosaic). arXiv:2411.03538, **NeurIPS 2024 Workshop on Adaptive Foundation Models** (workshop, not main track). 20 models, 2Kโ†’128K, text-embedding-3-large, 512-token chunks, FAISS. Accuracy 64k โ†’ 125k: **holds up** โ€” o1-preview 0.831โ†’0.763, GPT-4o 0.769โ†’0.767, Claude 3.5 Sonnet 0.741โ†’0.706; **degrades** โ€” Llama 3.1 405B 0.587โ†’0.426, GPT-4 Turbo 0.623โ†’0.560. The valuable part is that **failure modes are qualitatively distinct**: Claude 3 Sonnet refused on copyright grounds increasingly with length; Gemini 1.5 Pro tripped safety filters; DBRX summarized instead of answering above 16k; Mixtral emitted repeated nonsense; Llama 3.1 405B gave consistent wrong answers. โš ๏ธ Late-2024 model generation; no equally systematic 2026 redo exists. + +**Effective context length โ‰ช advertised.** +- **RULER** โ€” Hsieh, Sun, Kriman, Acharya, Rekesh, Jia, Zhang, Ginsburg (NVIDIA). arXiv:2404.06654, **COLM 2024**. All models claim โ‰ฅ32K; **only half maintain satisfactory performance at 32K**, despite near-perfect vanilla NIAH. +- **โญ NoLiMa** โ€” Modarressi, Deilamsalehy, Dernoncourt, Bui, Rossi, Yoon, Schรผtze (LMU + Adobe Research). arXiv:2502.05167, **ICML 2025**. Needles share **no lexical overlap** with the question โ€” only associative links. **13 models all claiming โ‰ฅ128K: at 32K tokens, 11 of 13 fall below 50% of their short-context baseline. GPT-4o: 99.3% โ†’ 69.7%.** Reasoning-enhanced models and CoT don't rescue it. **The cleanest demonstration that NIAH scores are an artifact of literal matching.** +- **HELMET** โ€” Yen, Gao, Hou, Ding, Fleischer, Izsak, Wasserblat, Chen (Princeton + Intel Labs). arXiv:2410.02694, **ICLR 2025**. Seven application-centric categories to 128k. **"Synthetic tasks like NIAH do not reliably predict downstream performance"**; categories show low correlation with each other; the open-vs-closed gap **widens with length**. +- **LongBench v2** โ€” arXiv:2412.15204, **ACL 2025**. 503 MCQs, 8kโ€“2M words. **Human experts under a 15-min limit: 53.7%. Best direct-answering model: 50.1%. o1-preview with extended reasoning: 57.7%.** +- **โญ Context Length Alone Hurts LLM Performance Despite Perfect Retrieval** โ€” Du, Tian, Ronanki, Rongali, Bodapati, Galstyan, Wells, Schwartz, Huerta, Peng. arXiv:2510.05381, **Findings of EMNLP 2025**. **Even with perfect retrieval, accuracy degrades 13.9%โ€“85% as input length grows โ€” and the degradation persists when irrelevant content is replaced with whitespace or masked entirely.** Length itself, independent of distraction and retrieval quality, is a failure axis. Mitigation: prompt the model to recite retrieved evidence before answering (+up to 4% for GPT-4o on RULER). +- โš ๏ธ **"Context Rot"** (Kelly Hong, Anton Troynikov, Jeff Huber, Chroma, Jul 2025) is a **blog/tech report, not peer-reviewed**. Cite Du et al. (Findings EMNLP 2025) instead for the same claim. + +**"Lost in the middle" has been substantially revised.** *Positional Biases Shift as Inputs Approach Context Window Limits* โ€” Veseli, Chibane, Toneva, Koller (Saarland/MPI-SWS). arXiv:2508.07479, **COLM 2025**. **The U-curve is strongest only when the input occupies up to ~50% of the context window.** Beyond that, primacy weakens, recency holds, and it becomes a **distance-based bias**. Measuring in *relative* rather than absolute length is the methodological point most prior work got wrong. Also: *Lost in the Middle: An Emergent Property from Information Retrieval Demands* (arXiv:2510.10276); *On the Emergence of Position Bias in Transformers* (arXiv:2502.01951); and arXiv:2511.05850 reporting **Gemini 2.5 Flash shows no lost-in-the-middle effect for simple factoid QA**. Counterweight: *Stable-RAG* (arXiv:2601.02993) shows retrieval-**permutation**-induced hallucinations remain measurable. + +**โญ The most replicated practical finding of 2025โ€“2026 is the dullest: preserve document order.** *Stronger Baselines for RAG with Long-Context LMs* โ€” Alex Laitenberger, **Christopher D. Manning**, Nelson F. Liu (Stanford). arXiv:2506.03989, **EMNLP 2025**. **DOS RAG** ("Document's Original Structure") = retrieve-then-read preserving original passage order. It **matches or outperforms ReadAgent and RAPTOR** across long-context QA benchmarks and systematically varied token budgets; recommended as the mandatory baseline for future RAG papers. Same insight as OP-RAG, independently arrived at by a different group, and peer-reviewed. + +**The cost evidence is unambiguous.** *The Token Tax of Epistemic Accuracy* โ€” Hamilton, Singh, Wise, Yousif, Carvalho, Shan, Mayyas, Cavuoto, Megahed. arXiv:2606.20898, 18 Jun 2026. 972 answers, expert-validated manufacturing-safety benchmark: **long-context 73.1% correctness vs 65.4% for semantic RAG โ€” at 26ร— the per-query token cost.** โš ๏ธ Small LMs, narrow domain; the gap would likely shrink with frontier models and a better-engineered RAG arm. Also: clinical EHR reasoning (arXiv:2508.14817) โ€” **RAG with <8K tokens matches long-context**; on-device unified compression matches traditional RAG at **1/10 the context** (arXiv:2604.14403). + +**And a strong result for minimal RAG:** *Frustratingly Simple Retrieval Improves Challenging, Reasoning-Intensive Benchmarks* โ€” Lyu, Duan, Shao, Koh, Min (UW/Berkeley/AI2). arXiv:2507.01297. A *minimal* RAG pipeline over CompactDS gives **+10% MMLU, +33% MMLU Pro, +14% GPQA, +19% MATH** across 8Bโ€“70B, matching or beating Google Search and agentic RAG. + +**โ†’ OPEN DEBATE #8 โ€” LC vs RAG.** Google DeepMind (EMNLP 2024) and Li et al. (arXiv:2501.01880) say LC wins on average. NVIDIA/OP-RAG and Lyu et al. say a properly ordered RAG beats LC at a fraction of the tokens. LaRA (arXiv:2502.09977, 2,326 test cases) says the question is ill-posed โ€” it depends on parameter size, long-text ability, context length, task type, and chunk characteristics. Hamilton et al. quantify it as +7.7 points for 26ร— cost. **The reconciliation the 2025โ€“26 evidence supports: most published "LC beats RAG" results used a weak RAG arm (relevance-ordered, badly chunked, small top-k). Fix ordering (DOS/OP-RAG) and recall and RAG closes most of the gap at 3โ€“25ร— lower cost. Meanwhile LC has an independent problem โ€” Du et al. show accuracy falls 13.9โ€“85% with length even under perfect retrieval with distractors literally blanked out โ€” so "just use the long window" is not stable as context grows.** Note also Li et al.'s under-quoted finding that **summarization-based retrieval performs comparably to LC while chunk-based retrieval lags** โ€” the RAG side of most comparisons is handicapped by chunking, not by retrieval. + +--- + +## 5. VERIFICATION, GROUNDING, ATTRIBUTION + +### 5.1 Claim-level attribution + +**The benchmarks.** *ALCE* (Tianyu Gao, Howard Yen, Jiatong Yu, Danqi Chen; **EMNLP 2023**, arXiv:2305.14627) โ€” citation recall + precision via a TRUE-style NLI model; **~50% of the best models' generations on ELI5 are not fully supported by their own cited passages**. *AttrScore* (Xiang Yue, Boshi Wang, Ziru Chen, Kai Zhang, Yu Su, Huan Sun; **Findings EMNLP 2023**, arXiv:2305.06311) โ€” defines the attributable/extrapolatory/contradictory taxonomy. โš ๏ธ **Per-model F1 numbers not extractable from the abstract or Anthology page โ€” do not quote a specific AttrScore F1.** *HAGRID* (Ehsan Kamalloo, Aref Jafari, Xinyu Zhang, Nandan Thakur, **Jimmy Lin**; arXiv:2307.16883). *AttributionBench* (arXiv:2402.15089, Findings ACL 2024, OSU NLP) โ€” โš ๏ธ from a search summary: **even a fine-tuned GPT-3.5 reaches only ~80% macro-F1** on binary supportedness. *LFRQA / RAG-QA Arena* (arXiv:2407.13998, EMNLP 2024, AWS AI Labs) โ€” 26K queries, 7 domains; โš ๏ธ frequently misdescribed as an attribution benchmark, it is an answer-quality arena. + +**โญ The single most important 2026 result: attribution evaluators do NOT transfer.** "Do LLM Attribution Metrics Transfer? Auditing RAG Evaluation Across Datasets and Constructs" โ€” Tianyu Ding, Aditya Nannapaneni, Juan Pablo De la Cruz Weinstein. arXiv:2606.23915, 22 Jun 2026. 8 scorers ร— 3 constructs; 1,610 AttributionBench + 2,150 HAGRID examples. + +- **No scorer stayed inside the 95% CI across all datasets within any construct.** +- **Metric rankings invert across datasets: Kendall tau = โˆ’0.64, p = 0.031.** An actual reversal, not noise. +- **One NLI scorer: AUROC 0.90 on short-claim AttributedQA โ†’ 0.53 (chance) on long-form LFQA.** The standard NLI citation checker used by ALCE-style pipelines is **at chance on long-form answers**. +- **BERTScore was the best scorer on LFQA at 0.91 AUROC** โ€” the "dumb" metric beat entailment models exactly where entailment should shine. +- Selecting a scorer by average cross-dataset performance: **mean regret 0.172 AUROC** under leave-one-dataset-out. +- LLM judges avoid total collapses but cost **~100ร—** and are non-deterministic. + +โš ๏ธ Single preprint, not peer-reviewed; 8 scorers is decent but not exhaustive. Still the strongest available evidence that **any single reported attribution-evaluator score is a property of the dataset, not the evaluator.** + +**โญ Citation quality in deployed deep-research agents.** "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents" โ€” Hailey Onweller, Elias Lumer, Austin Huber, Pia Ramchandani, Vamse Kumar Subbiah, Corey Feld. arXiv:2605.06635, 7 May 2026. +- Frontier models: **>94% link accessibility, >80% topical relevance, but only 39โ€“77% of citations are factually accurate against the source content.** +- **Fewer than half of open-source models can produce a cited report at all** one-shot. +- **Fact-check accuracy drops ~42% on average across two frontier models as tool calls scale from 2 to 150.** Longer research trajectories, monotonically worse grounding. + +โš ๏ธ Preprint; the fact-check judge is itself an LLM (so ยง5.1's transfer problem applies to this paper's own instrument); "two frontier models" is thin. But *links resolve, topics match, claims don't* is the most actionable attribution finding of 2026 โ€” and the tool-call scaling result is directly relevant to agentic retrieval design. + +**Does citation generation degrade answer quality? Yes, and granularity is the knob.** "Are Finer Citations Always Better? Rethinking Granularity for Attributed Generation" โ€” Hexuan Wang, Jingyu Zhang, **Benjamin Van Durme, Daniel Khashabi** (JHU). arXiv:2604.01432, Apr 2026. **Forcing sentence-level citations degrades attribution quality by 16โ€“276% relative to the optimal granularity**, across model sizes; quality **peaks at intermediate (paragraph-level) granularity**; **the penalty for fine-grained constraints grows with model scale**. At the right granularity you get both attribution and correctness. Proposed mechanism: attention dilution during synthesis plus atomic units fracturing the semantic context needed to synthesize. + +### 5.2 Hallucination detection โ€” the numbers + +**Benchmarks.** *RAGTruth* (Cheng Niu, Yuanhao Wu, Juno Zhu, Siliang Xu, Kashun Shum, Randy Zhong, Juntong Song, Tong Zhang; **ACL 2024**, arXiv:2401.00396) โ€” **~18,000 naturally generated RAG responses with word-level manual hallucination annotations** from 6 LLMs across QA / data-to-text / summarization. Still the de facto standard; โš ๏ธ generators are 2023-vintage, a real staleness concern for 2026. *HaluBench* (via Lynx) โ€” 15,000 samples across HaluEval, DROP, CovidQA, PubMedQA, FinanceBench, RAGTruth; human validation agreement **0.90โ€“0.96**; โš ๏ธ constructed partly by **semantic perturbation**, a distribution unlike naturally-occurring hallucination. *FaithBench* (**NAACL 2025 short**, Vectara-affiliated) โ€” deliberately built from summaries **where SOTA detectors including GPT-4o-as-judge disagree**; **most SOTA detectors score near 50% (chance)**; โš ๏ธ adversarial by construction, so ~50% is partly definitional, and the team also ships a competing detector. + +**โญ TRIVIA+ / the 2026 benchmark-hygiene paper.** "Rethinking Evaluation for LLM Hallucination Detection: A Desiderata, A New RAG-based Benchmark, New Insights" โ€” Wenbo Chen, Veena Padmanabhan, Tootiya Giyahchi, Elaine Wong, Leman Akoglu. arXiv:2605.11330, 11 May 2026. RAG-based with **"the longest context in the literature"** plus **four synthetic label-noise sets**. Three findings: current detectors have ample room on RAG tasks; **plain LLM-as-a-Judge is competitive with specialized detectors**; **label noise degrades detection and shifts detector rankings unpredictably**. โš ๏ธ **I could not extract the AUROC table** (PDF compressed streams) โ€” qualitative findings only. + +**Detector numbers I can stand behind.** *LettuceDetect* โ€” รdรกm Kovรกcs, Gรกbor Recski (KR Labs / TU Wien). arXiv:2502.17125, 24 Feb 2025. ModernBERT-based, 8k context, token-level classification. RAGTruth example-level F1: + +| System | F1 | +|---|---| +| GPT-4 (prompt-based) | 63.4% | +| Luna (Galileo encoder) | 65.4% | +| **LettuceDetect-large** | **79.22%** | +| Fine-tuned Llama-2-13B | 78.7% | +| **Fine-tuned Llama-3-8B (prior SOTA)** | **83.9%** | + +**LettuceDetect is NOT SOTA on RAGTruth โ€” fine-tuned Llama-3-8B at 83.9% is.** LettuceDetect is the best *cost-adjusted* detector: ~30ร— smaller, 30โ€“60 examples/sec on one GPU, +14.8% relative over Luna. The paper itself says "competitive with." The widely-repeated "LettuceDetect is SOTA" framing is wrong. + +*Lynx* โ€” Selvan Sunitha Ravi, Bartosz Mielczarek, Anand Kannappan, **Douwe Kiela**, Rebecca Qian (Patronus AI). arXiv:2407.08488. HaluBench accuracy: + +| Model | Overall | HaluEval | DROP | CovidQA | PubMedQA | FinanceBench | RAGTruth | +|---|---|---|---|---|---|---|---| +| **Lynx 70B** | **87.4%** | 88.4 | 86.4 | 97.5 | 90.4 | 81.4 | 80.2 | +| GPT-4o | 86.5% | 87.9 | 84.3 | 95.0 | 82.1 | **85.3** | **84.3** | +| Lynx 8B | 82.9% | 85.7 | 77.8 | 96.3 | 85.2 | 72.5 | 80.0 | +| Claude-3-Sonnet | 78.8% | 84.5 | 84.3 | 95.0 | 82.9 | 69.7 | 79.1 | +| RAGAS Faithfulness | 66.9% | โ€” | โ€” | โ€” | โ€” | โ€” | โ€” | + +โš ๏ธ **The marketing omits that the 87.4 vs 86.5 win is 0.9 points, and GPT-4o beats Lynx-70B on FinanceBench (85.3 vs 81.4) and RAGTruth (84.3 vs 80.2).** Lynx's margin comes almost entirely from PubMedQA and CovidQA โ€” a benchmark-composition win. Vendor-authored, vendor-constructed benchmark, perturbation-synthesized hallucinations. "Lynx 2.0" is blog-announced with no paper found. + +**Vectara HHEM leaderboard โ€” vendor-run, flagged.** Fetched github.com/vectara/hallucination-leaderboard. **Last updated 11 May 2026; current scoring model HHEM-2.3** (not 2.1). Methodology: **>7,700 source articles, 50 to 24,000 words**, across news/tech/science/medicine/legal/sports/business/education; instruction "Summarize using only the information in the given passage. Do not infer"; **temperature 0**; reports hallucination rate, factual consistency, and **answer rate**. Top as of 11 May 2026: antgroup/finix_s1_32b **1.8%** (99.5% answer rate); openai/gpt-5.4-nano **3.1%** (100%); google/gemini-2.5-flash-lite **3.3%**; microsoft/Phi-4 **3.7%** but only **80.7% answer rate**; meta-llama/Llama-3.3-70B **4.1%**. + +โš ๏ธ Caveats that matter: it is **graded by the vendor's own product**, so it is simultaneously a benchmark and a product demo; it measures **summarization faithfulness only**; the **answer-rate column is the tell** (3.7% at 80.7% answer rate is not comparable to 3.1% at 100%); a 32B model from a payments company topping a board of frontier models is the pattern you'd expect from optimization-toward-the-metric. Vectara's own repo text: determining hallucinations is "impossible without a reference source," and the problem "is far from solved." โš ๏ธ Figures circulating in SEO summaries (GPT-5 Pro ~1.0%, Claude Opus 4.7 ~1.2%) **do not appear on the leaderboard I fetched โ€” do not use them.** + +**Semantic entropy and its 2026 pressure.** *SelfCheckGPT* (Manakul, Liusie, Gales; EMNLP 2023) is now used almost exclusively as a baseline; the standard criticism is that surface-token divergence conflates paraphrase with contradiction. *Semantic entropy* (Farquhar, Kossen, Kuhn, Gal, **Nature 2024**) clusters samples by bidirectional entailment into meaning classes. *Semantic Entropy Probes* (Kossen et al., arXiv:2406.15927) approximate it in a **single forward pass**. + +2026 pressure (โš ๏ธ all from an arXiv listing fetch, IDs as returned, abstracts not individually opened): arXiv:2607.16868 โ€” **+7.1% AUROC over semantic entropy** via logical graphs, arguing entailment clustering is the wrong equivalence relation; arXiv:2606.10198 โ€” **+5โ€“20 points** via geometry; arXiv:2605.04295 โ€” **0.88 vs 0.65 AUROC** on TriviaQA with adaptive conformal SE; **arXiv:2605.05166 "The First Token Knows"** โ€” **first-token confidence matches or modestly exceeds semantic self-consistency with zero sampling overhead** (the sharpest efficiency critique); arXiv:2607.07670 โ€” 5-sample SE reaches only **0.71โ€“0.83 AUROC at 5ร— inference cost**, losing to activation probes; arXiv:2603.22812 โ€” **~50% fewer samples** via variance-based early termination; arXiv:2606.24115 โ€” clustering underperforms plain token statistics in a VLM medical domain. + +**Honest synthesis:** nobody has published a clean refutation, and the *idea* (uncertainty over meanings, not tokens) is not seriously contested. What is contested is (a) whether the cost is justified vs single-pass probes, (b) whether bidirectional-entailment clustering is the right equivalence relation, (c) whether it transfers across domains. Also: **it is gray-box** โ€” it needs token probabilities, unusable behind most commercial APIs. + +**Probing / internal states.** *LLMs Know More Than They Show* โ€” Hadas Orgad, Michael Toker, Zorik Gekhman, Roi Reichart, Idan Szpektor, Hadas Kotek, **Yonatan Belinkov**. **ICLR 2025**, arXiv:2410.02707. Four findings, and the third is the one people skip: (1) internal representations encode more truthfulness signal than thought; (2) the signal is **concentrated in specific tokens**; (3) **error detectors fail to generalize across datasets โ€” truthfulness encoding is not universal but multifaceted**; (4) models can internally encode the right answer and still emit the wrong one. Counter-current: arXiv:2510.09033 argues the probe signal is largely **recall/familiarity, not truthfulness**. โš ๏ธ Active dispute; both cannot be right in their strong forms. + +*ReDeEP* (Zhongxiang Sun, Xiaoxue Zang, Kai Zheng, Yang Song, Jun Xu, Xiao Zhang, Weijie Yu, Han Li; arXiv:2410.11414) โ€” mechanistic claim that RAG hallucination occurs when **Knowledge FFNs over-weight parametric knowledge in the residual stream while Copying Heads fail to integrate retrieved context**. โš ๏ธ **The abstract carries no numbers**; a third-party 2026 paper (arXiv:2605.07209) reports **ReDeEP token-level AUC 0.73 on RAGTruth**, beating it by 7.4โ€“10.3 points โ€” treat as unverified secondary reporting. + +**Which methods generalize? None of them, cleanly.** Three independent results converge: attribution scorers invert rankings across datasets (ฯ„ = โˆ’0.64) and one drops to chance; internal-state error detectors fail to generalize (Orgad et al., ICLR 2025); detector rankings shift unpredictably under label noise and plain LLM-as-a-Judge is competitive (TRIVIA+). **The practical implication is uniform: validate a detector on your own target distribution. Picking by published averages carries ~0.17 AUROC mean regret. And the honest baselines to beat are LLM-as-a-Judge and BERTScore, not SelfCheckGPT.** + +### 5.3 Do verification passes measurably help? + +**The negative camp.** *Large Language Models Cannot Self-Correct Reasoning Yet* โ€” Jie Huang, Xinyun Chen, Swaroop Mishra, Huaixiu Steven Zheng, Adams Wei Yu, Xinying Song, **Denny Zhou** (Google DeepMind/Research). **ICLR 2024**, arXiv:2310.01798. **"LLMs struggle to self-correct their responses without external feedback, and at times, their performance even degrades after self-correction."** The methodological contribution is showing prior reported gains came from **oracle labels leaking in** (the loop was told when to stop) or from baselines not given equal compute. + +*On the Self-Verification Limitations of LLMs on Reasoning and Planning Tasks* โ€” Kaya Stechly, Karthik Valmeekam, **Subbarao Kambhampati**. **ICLR 2025**. โš ๏ธ Search returned arXiv:2402.08115; I could not verify that ID. Finding: LLM self-critique does not deliver iterative improvement; sound *external* verifiers do. + +*SELF-[IN]CORRECT* (arXiv:2404.04298) gives the mechanism: models are no better at **discriminating** among their own candidates than at generating a good one. If generation and discrimination are equally strong, a self-critique loop has no information advantage. + +**โญ *Feedback Friction: LLMs Struggle to Fully Incorporate External Feedback*** โ€” Dongwei Jiang, Alvin Zhang, Andrew Wang, Nicholas Andrews, **Daniel Khashabi**. arXiv:2506.11930, v2 21 Sep 2025. Uncomfortable for the "just add external feedback" fix: **even under near-perfect, complete external feedback, models plateau below the achievable ceiling** across math, knowledge, scientific, and multi-domain reasoning including Claude 3.7 with extended thinking. Progressive temperature sampling and explicit rejection of prior wrong answers **still fail**. **Model confidence, measured via semantic entropy, predicts feedback resistance** โ€” high-confidence wrong answers are the ones that won't budge. โš ๏ธ No numeric ceilings in the abstract. + +*Decomposing LLM Self-Correction: The Accuracy-Correction Paradox and Error Depth Hypothesis* (arXiv:2601.00828, Jan 2026) โ€” **weaker models show 1.6ร— higher intrinsic correction rates than stronger models**. Strong models' remaining errors are *deep* errors self-correction structurally cannot reach, so raw "correction rate" is inversely correlated with base capability and is an actively misleading metric. + +**The positive camp.** *CorrectBench* (Guiyao Tie et al., arXiv:2510.16062) โ€” self-correction does improve accuracy, especially on complex reasoning. But three quiet negatives inside: **reasoning models (DeepSeek-R1) show limited additional gain at high time cost** (RL-trained reasoning already internalizes the loop); combining strategies helps at an efficiency cost; and **plain CoT is competitive on accuracy *and* efficiency.** + +**โญ The strongest RAG-specific positive, and it credits the verifier not the reflection.** *Self-Correcting RAG (NLI-guided MCTS)* โ€” Shijia Xu, Zhou Wu, Xiaolong Jia, Yu Wang, Kai Liu, April Xiaowen Dong. arXiv:2604.10734, 12 Apr 2026: + +| Metric | Standard RAG | Self-Correcting RAG | +|---|---|---| +| Attribution Precision | 0.52 | **0.85** | +| Contradiction Rate | 0.15 | **0.04** | +| Supportability | 0.65 | **0.88** | +| EM (avg, 6 datasets) | 25.8% | **37.1%** | +| F1 (avg, 6 datasets) | 36.1% | **45.8%** | + +**The ablation is load-bearing: MMKP context selection alone gives EM 34.5% but attribution precision only 0.58. NLI-guided MCTS alone gives attribution precision 0.82 at EM 31.2%.** Better context alone barely moves faithfulness; **the verifier does essentially all the faithfulness work.** Datasets: NQ, PopQA, MuSiQue, 2Wiki, HotpotQA (1,000 queries each) + MultiHop-RAG (2,556). โš ๏ธ Preprint; "Attribution Precision" is itself an automatic scorer, which ยง5.1 says may not transfer. + +*CRAG* (Yan et al., **ICLR 2024**, arXiv:2401.15884) โ€” lightweight T5 retrieval evaluator triggering correct/ambiguous/incorrect actions with web-search fallback. **2026 reproduction** (arXiv:2603.16169) finds two things: (a) the original relied on the Google Search API and closed weights; swapping in Wikipedia API + Phi-3-mini **reproduces comparable performance**; (b) **SHAP analysis shows CRAG's T5 evaluator primarily keys on named-entity alignment, not semantic similarity** โ€” the "retrieval quality evaluator" is substantially a lexical entity-overlap detector. A meaningful deflation even though the numbers replicate. *Self-RAG* (Asai et al., ICLR 2024) trains the critique in via reflection tokens; โš ๏ธ **I found no rigorous 2025โ€“26 independent reproduction.** + +**โ†’ THE HONEST NET VERDICT.** The apparent conflict dissolves along one axis: +1. **Intrinsic self-correction โ€” no external signal, no oracle stopping โ€” does not reliably help and sometimes hurts.** Huang et al. (ICLR 2024) canonical; Stechly/Kambhampati extends to planning; SELF-[IN]CORRECT gives the mechanism; the 2026 error-depth paper shows the apparent gains shrink as base models strengthen. **Nothing in 2025โ€“26 overturns this.** +2. **Verification against an external, sound signal does help substantially in RAG.** The NLI-MCTS ablation is the cleanest demonstration: verifier alone moves attribution precision 0.52 โ†’ 0.82; better retrieval alone moves it 0.52 โ†’ 0.58. **The gain is in the verifier, not in the reflection.** +3. **But external feedback is not a solved fix.** Feedback Friction: models resist even near-perfect feedback, and semantic-entropy confidence predicts which errors are unfixable. +4. **If you are running a modern reasoning model, a bolted-on critique pass is likely a latency tax** (CorrectBench). +5. **Verification-loop gains are hard to measure honestly**, because the faithfulness metrics used to score them are the same ones ยง5.1 showed don't transfer. + +**Practical distillation: add a verification pass only when it consumes something the generator did not already see** โ€” retrieved evidence checked by an independent NLI model, a search result, an executor, a sound checker. A pass that re-reads the model's own output and asks "are you sure?" is a latency tax. + +### 5.4 Groundedness benchmarks in 2026 + +**FACTS Benchmark Suite (Google DeepMind + Kaggle), released 9 Dec 2025.** Supersedes the standalone FACTS Grounding leaderboard by absorbing it. Four pillars: Parametric (2,104), Search (1,884, standardized web-search tool, often multi-hop), Multimodal (1,522), Grounding v2. **3,513 examples publicly released** (โš ๏ธ the four listed sizes exceed this โ€” presumably the public split only; I could not resolve the inconsistency). **Kaggle owns the private held-out sets, runs the evaluations, and hosts the leaderboard** โ€” the right structural answer to contamination and the main reason to cite FACTS in 2026. + +**Results: 15 leading models tested. Gemini 3 Pro leads at 68.8% FACTS Score. Every model โ€” Claude, GPT, Llama included โ€” scores below 70%.** + +โš ๏ธ **Google builds the benchmark and Google's model tops it.** Kaggle's custody of the private sets mitigates contamination but not design bias. The earlier FACTS Grounding used an **ensemble of three frontier judges** (Gemini 1.5 Pro, GPT-4o, Claude 3.5 Sonnet) to dilute single-judge bias; the blog does not state whether the Suite keeps that ensemble. โš ๏ธ Older grounding-only figures circulating (Gemini 2.0 Flash Exp 83.6%, Claude 3.5 Sonnet 79.4%, GPT-4o 78.8%) come from a vendor blog, not DeepMind, and I could not confirm them against Kaggle (page rendered empty). Note the ~80% grounding-only and ~69% composite are **different scales measuring different things** โ€” not a trend. + +โš ๏ธ 2026 entrants surfaced but not fetched: **LayerRAG-Bench** (arXiv:2607.27353) โ€” argues groundedness-only evaluation produces **false positives** when answers look grounded but fail at the evidence, tool-contract, authorization, or session-state layer (the right critique for agentic RAG); **From Binary Groundedness to Support Relations** (Sarkar, Poelitz, Kewenig, **Microsoft Research**, arXiv:2604.08082) โ€” argues the binary supported/unsupported framing is wrong. + +**Where frontier models actually sit, in one paragraph.** On the most rigorous contamination-controlled composite (FACTS Suite, private held-out sets): **no model exceeds 70%; the leader is 68.8%.** On grounding-in-provided-documents, the best models sit around 80โ€“84% (older leaderboard). On summarization faithfulness with an explicit "do not infer" instruction at temp 0, the best hallucinate on **~2โ€“4%** of summaries (vendor-scored). On RAG QA faithfulness, error rates are several times higher. **On citation-level factual support in deep research agents, only 39โ€“77% of citations actually support their claim, degrading ~42% as tool calls scale from 2 to 150.** The spread between those numbers *is* the story: the more constrained the grounding task, the better models look, and **the number that matters most for production agentic RAG is the worst one**. + +--- + +## 6. GRAPHRAG โ€” DID THE EVIDENCE HOLD UP? + +### 6.1 The origin paper never got a venue + +"From Local to Global: A Graph RAG Approach to Query-Focused Summarization" โ€” Darren Edge, Ha Trinh, Newman Cheng, Joshua Bradley, Alex Chao, Apurva Mody, Steven Truitt, Dasha Metropolitansky, Robert Osazuwa Ness, Jonathan Larson (**Microsoft Research**). arXiv:2404.16130, v1 24 Apr 2024, v2 19 Feb 2025. **1,875 citations.** + +Two corpora โ€” podcast transcripts (~1M tokens, 8,564 entities, 20,691 edges) and news (~1.7M tokens, 15,754 entities, 19,520 edges). **72โ€“83% comprehensiveness win rates and 62โ€“82% diversity win rates vs naive vector RAG**, judged pairwise by an LLM. + +โš ๏ธ **I could not verify any peer-reviewed venue.** No journal-ref or venue comment on arXiv through v2 (Feb 2025). As best I can establish, **the single most influential GraphRAG paper remained an arXiv preprint / MSR tech report and was never published at ACL/EMNLP/NeurIPS/ICLR.** The field built on an unrefereed baseline. And the numbers are **LLM-as-judge pairwise preferences on comprehensiveness and diversity** โ€” not accuracy, not F1, not human evaluation โ€” over **LLM-generated evaluation questions**, with no cost figures. + +### 6.2 The successor line + +| System | Paper | Venue | Headline | +|---|---|---|---| +| **RAPTOR** | Sarthi, Abdullah, Tuli, Khanna, Goldie, **Manning** (Stanford), arXiv:2401.18059 | **ICLR 2024** | **+20% on QuALITY** with GPT-4; +2% over DPR, +5.1% over BM25 | +| **LightRAG** | Zirui Guo, Lianghao Xia, Yanhua Yu, Tu Ao, Chao Huang (HKU), arXiv:2410.05779 | **EMNLP 2025** | Dual-level graph retrieval; lower cost + faster incremental update than GraphRAG | +| **HippoRAG 2** | Bernal Jimรฉnez Gutiรฉrrez, Yiheng Shu, Weijian Qi, Sizhe Zhou, Yu Su (Ohio State), arXiv:2502.14802 | **ICML 2025** | **+7% associative memory** over SOTA embeddings. **The abstract itself concedes prior graph methods' "performance on more basic factual memory tasks drops considerably below standard RAG"** | +| **GraphReader** | arXiv:2406.14550 | **Findings EMNLP 2024** | 4k-context GraphReader beats GPT-4-128k on LV-Eval across 16kโ€“256k | +| **Think-on-Graph 2.0** | arXiv:2407.10805 | **ICLR 2025** | +5.51% over CoK on HotpotQA | +| **KAG** | Ant Group / OpenSPG, arXiv:2409.13731 | **ACM Web Conf 2025** | Schema-constrained KG to fix OpenIE noise in HippoRAG/GraphRAG | +| **PathRAG** | arXiv:2502.14902 | โš ๏ธ venue unverified | Flow-based path pruning. Key framing: **the problem with GraphRAG is redundancy, not insufficiency** | +| **Youtu-GraphRAG** | Junnan Dong, Siyu An, โ€ฆ Xiao Huang, Yunsheng Wu, Di Yin, Xing Sun (Tencent Youtu + Monash + HK PolyU), arXiv:2508.19855 | โš ๏ธ GitHub says ICLR 2026; arXiv comments do not | **+16.62% accuracy, โˆ’90.71% token cost** in graph construction | +| **KET-RAG** | arXiv:2502.09304 | **KDD 2025** | **20% indexing cost reduction** | +| **LazyGraphRAG** | Microsoft Research, 25 Nov 2024 | โš ๏ธ **blog, not a paper** | Indexing cost **identical to vector RAG, 0.1% of full GraphRAG**; comparable global-query quality at **>700ร— lower query cost** | + +That last row is the tell. **You don't build LazyGraphRAG unless the original was unaffordable.** + +### 6.3 The skeptical line โ€” this is the substantive part + +**โญ RAG vs. GraphRAG: A Systematic Evaluation and Key Insights** โ€” Haoyu Han, Li Ma, Yu Wang, Harry Shomer, Yongjia Lei, Zhisheng Qi, Kai Guo, Zhigang Hua, Bo Long, Hui Liu, **Charu C. Aggarwal**, Jiliang Tang (**Michigan State + Meta + IBM** โ€” *not* NVIDIA; I found no NVIDIA study of this name). arXiv:2502.11371, v1 17 Feb 2025, **v3 4 Mar 2026.** + +- **Natural Questions (single-hop):** RAG **64.78 F1** > Community-GraphRAG(Local) 63.01 > HippoRAG2 61.03. **Plain RAG wins.** +- **HotpotQA:** HippoRAG2 63.01 > Community-GraphRAG(Local) 61.66 > RAG 60.04. +- **MultiHop-RAG (accuracy):** HippoRAG2 **70.27** > Community-GraphRAG(Local) 69.01 > RAG 67.02. **The multi-hop win is ~+3 points, not a category change.** +- **Complementarity: 13.6% of MultiHop-RAG queries are GraphRAG-only wins, 11.6% are RAG-only wins** โ€” nearly symmetric. +- **Summarization (SQuALITY, ROUGE-2 F1): RAG 10.08, Community-GraphRAG(Local) 10.10, Community-GraphRAG(Global) 6.99.** The global/community mode โ€” Microsoft GraphRAG's entire selling point โ€” **loses badly on a real summarization metric.** +- **Cost (MultiHop-RAG):** construction time RAG **135s** vs KG-GraphRAG **7,702s** vs Community-GraphRAG **5,560s** โ€” a **41โ€“57ร— indexing penalty**. Retrieval latency 1,724s / 14,434s / 1,249s. Storage 127 / 117 / 165 MB. +- Hybrid integration: **+6.4%** on MultiHop-RAG with Llama-3.1-70B. +- **Methodological warning: "position bias is clearly present in LLM-as-a-Judge evaluations"** โ€” directly undercutting Edge et al.'s win-rate methodology. + +**โญ When to use Graphs in RAG / GraphRAG-Bench** โ€” Zhishang Xiang, Chuanjie Wu, Qinggang Zhang, Shengyuan Chen, Zijin Hong, Xiao Huang, Jinsong Su. arXiv:2506.05690, v1 6 Jun 2025, **v3 22 Feb 2026**. โš ๏ธ GitHub labels it ICLR 2026; unverified from arXiv. The abstract opens with the negative framing: *"recent studies report that GraphRAG frequently underperforms vanilla RAG on many real-world tasks."* + +Across 7 frameworks (MS-GraphRAG, HippoRAG, HippoRAG2, LightRAG, Fast-GraphRAG, RAPTOR, Lazy-GraphRAG): +- **Level 1 fact retrieval, Novel:** Basic RAG w/ rerank **60.92%** vs best GraphRAG (HippoRAG2) 60.14% โ€” *graph loses.* +- **Level 2 complex reasoning, Novel:** Basic RAG 42.93% โ†’ HippoRAG2 **53.38% (+24.4% relative)** โ€” the clearest genuine graph win. +- **Level 3 contextual summarization, Novel:** Basic RAG 51.30% โ†’ MS-GraphRAG **64.40%**. +- **Evidence recall inverts:** on fact retrieval, Basic RAG recall **83.21%** vs HippoRAG2 70.29%. Graph retrieval is *worse at finding the right evidence* for simple questions. +- **The cost table is the headline โ€” average prompt tokens per query:** + +| Method | Novel | Medical | +|---|---|---| +| Vanilla RAG | **879** | **954** | +| HippoRAG2 | 1,008 | 1,020 | +| RAPTOR | 3,441 | 3,510 | +| HippoRAG | 7,208 | 7,342 | +| LightRAG | 100,832 | 100,310 | +| **MS-GraphRAG (global)** | **331,375** | 332,881 | + +**Microsoft GraphRAG global search burns ~377ร— the prompt tokens of vanilla RAG per query** โ€” inference cost, on top of indexing cost. Conclusion: graph "introduces redundant information, which in turn degrades context relevance." + +Also: **GraphRAG-Bench dataset paper** (Yilin Xiao, Junnan Dong, Chuang Zhou, Su Dong, Qian-wen Zhang, Di Yin, Xing Sun, Xiao Huang; arXiv:2506.02404) โ€” college-level questions across 16 disciplines / 20 textbooks, 9 methods, scoring reasoning coherence not just final answers. + +**โญ Do We Still Need GraphRAG? Benchmarking RAG and GraphRAG for Agentic Search Systems** โ€” Dongzhe Fan, Zheyi Xue, Siyuan Liu, Qiaoyu Tan. arXiv:2604.09666, **1 Apr 2026.** The 2026 reframing: does agentic multi-round retrieval make explicit graphs redundant? Standardized LLM backbone, retrieval budget, inference protocol, full test sets. +- **Single-shot:** general QA โ€” graph gives only **+0.47 avg**. Multi-hop โ€” graph gives **+27.23 avg** (Contain-EM). +- **With training-free agentic search:** dense RAG + GraphSearch **narrows the multi-hop gap by 32.3%**. +- **With GRPO agentic search:** dense RAG becomes best on Natural Questions; graph backends still lead multi-hop. +- Verdict: agentic search "substantially improves dense RAG and narrows the gap... Nevertheless, GraphRAG remains advantageous for complex multi-hop reasoning... **when its offline cost is amortized.**" + +**โญ The harshest result: BM25 Wins at Scale: A Scaling Study of RAG Paradigms** โ€” Pengyu Wang, Benfeng Xu, Shaohan Wang, Mingxuan Du, Xin Zeng, Huarui Wu, Lei Zhang, Licheng Zhang. arXiv:2607.26497, **29 Jul 2026** (v3 31 Jul). 28 strictly nested corpus tiers spanning ~450ร— expansion, fixed questions and reference docs. +- A file-system agent leads at small scale but **costs 39ร— more query tokens** at the largest tier. +- **BM25 overtakes it around 10M corpus tokens and leads at every larger shared tier, with a margin approaching 20 points at full scale.** +- **Graph-based RAG "encounters construction walls before deployment scale, and its scalable variants remain below BM25 at shared tiers."** At real corpus sizes GraphRAG cannot even be built, and the cheap variants lose to lexical search from 1994. +- Conclusion: "lexical retrieval is the strongest scalable default, while agentic reasoning works best **after** ranked discovery rather than in place of it." + +โš ๏ธ I searched for a paper titled "GraphRAG under fire" and **found no such academic paper.** Not asserting it exists. + +### 6.4 Honest 2026 verdict + +**The evidence did not stall โ€” it substantially deflated and re-scoped.** + +1. **Single-hop / factual lookup: graph loses.** 64.78 vs 63.01 F1 (NQ); 60.92% vs 60.14% (GraphRAG-Bench). Consistent across independent groups. The HippoRAG 2 authors concede it in their own abstract. +2. **Multi-hop: graph genuinely wins, but the magnitude is wildly contested โ€” +3 points (controlled same-backbone, MultiHop-RAG) to +24% relative (GraphRAG-Bench) to +27 EM (RAGSearch single-shot).** The spread *is* the finding: it depends almost entirely on whether the vector baseline is well-tuned and whether the questions are natively bridge-entity questions. +3. **Global/sensemaking summarization: the original claim is the least replicated.** Edge et al. reported 72โ€“83% comprehensiveness win rates on LLM-judge preference; the controlled ROUGE-2 replication gives Community-GraphRAG Global **6.99 vs RAG 10.08** โ€” global search *loses*. GraphRAG-Bench does find graph wins on Level-3 summarization. Verdict: **graph helps on breadth/coverage summarization, but the specific "global search over community summaries" mechanism is not well supported, and the original evidence rested on an LLM-judge protocol later shown to have position bias.** +4. **Cost is the decisive variable, not accuracy.** Indexing 41โ€“57ร— (135s โ†’ 5,560โ€“7,702s); inference **331k prompt tokens/query vs 879** (~377ร—). Microsoft's own LazyGraphRAG (0.1% indexing, >700ร— lower query cost), Youtu's โˆ’90.71% construction tokens, and KET-RAG's โˆ’20% are the same admission from three different groups. +5. **The 2026 threat isn't better vector RAG โ€” it's agentic search and BM25.** Agentic retrieval closes 32.3% of the multi-hop gap with no graph; at deployment scale, graph can't be constructed at all. +6. **The rule the literature converges on:** build a graph only when (a) queries are genuinely multi-hop over bridge entities, (b) the corpus is small/static enough to amortize indexing, and (c) you have already lost to a hybrid BM25+dense+reranker baseline. The 13.6%/11.6% complementarity says the right answer is **hybrid routing**, not graph-everything. + +--- + +## 7. MULTI-TURN / MEMORY-AUGMENTED RETRIEVAL AGENTS + +### 7.1 The load-bearing context paper + +**LLMs Get Lost In Multi-Turn Conversation** โ€” Philippe Laban, Hiroaki Hayashi, Yingbo Zhou, Jennifer Neville (**Microsoft Research + Salesforce Research**). arXiv:2505.06120, 9 May 2025. **378 citations.** + +- **Average โˆ’39% performance across six generation tasks**, single-turn โ†’ multi-turn, for *every* top open- and closed-weight LLM tested. **200,000+ simulated conversations.** +- **The decomposition is the important part: the drop is a minor loss in aptitude and a large increase in unreliability.** Models aren't dumber multi-turn โ€” they're erratic. +- Mechanism: LLMs "make assumptions in early turns and prematurely attempt to generate final solutions, on which they overly rely." + +**Why this matters for memory systems:** multi-turn failure is substantially a *generation-side* pathology, not purely a retrieval/recall pathology. A memory layer that only fixes recall addresses part of the problem. Papers claiming large multi-turn wins from memory alone should be read against this. + +### 7.2 Benchmarks + +**LongMemEval** โ€” Di Wu, Hongwei Wang, Wenhao Yu, Yuwei Zhang, Kai-Wei Chang, Dong Yu (UCLA / Tencent AI Lab / UCSD). arXiv:2410.10813, **ICLR 2025** (confirmed in arXiv comments + proceedings). 500 curated questions, freely scalable histories; five abilities including **knowledge updates and abstention**. **30% accuracy drop** for commercial assistants and long-context LLMs on sustained interaction. + +**LongMemEval-V2** โ€” Di Wu, Zixiang Ji, Asmi Kawatkar, Bryan Kwan, Jia-Chen Gu, Nanyun Peng, Kai-Wei Chang. arXiv:2605.12493, **12 May 2026**. 451 questions, up to **500 trajectories / 115M tokens**. **AgentRunbook-C 72.5%; AgentRunbook-R (RAG-based memory) 48.5%; off-the-shelf coding agent baseline 69.3%.** Note: **the RAG-memory approach loses to no memory system at all by 21 points.** + +**LoCoMo** โ€” Adyasha Maharana, Dong-Ho Lee, Sergey Tulyakov, Mohit Bansal, Francesco Barbieri, Yuwei Fang. **ACL 2024**, arXiv:2402.17753. ~600 turns / 16K tokens avg over up to 32 sessions. **This is the benchmark the entire industry memory-scoring war is fought on, and it is small and partly defective โ€” see ยง7.4.** + +**MemoryAgentBench** โ€” Yuanzhe Hu, Yu Wang, Julian McAuley (UCSD). arXiv:2507.05257, latest 28 Jun 2026. Four competencies: accurate retrieval, test-time learning, long-range understanding, **conflict resolution / selective forgetting**. "Current methods fall short of mastering all four." Memory agents on GPT-4o reach only **~60% on single-hop conflict resolution.** + +**MTRAG** โ€” Yannis Katsis, Sara Rosenthal, Kshitij Fadnis, Chulaka Gunasekara, Young-Suk Lee, Lucian Popa, Vraj Shah, Huaiyu Zhu, Danish Contractor, Marina Danilevsky (**IBM Research**). arXiv:2501.03468. โš ๏ธ IBM's publications page lists it as **TACL**; arXiv v1 has no venue comment. **110 fully human-generated conversations, avg 7.7 turns, 4 domains, 842 tasks.** SOTA RAG systems struggle specifically on **later turns, unanswerable questions, and non-standalone questions**. Now the basis of **SemEval-2026 Task 8 (MTRAGEval)**. + +**HELMET** (**ICLR 2025**) covers multi-turn categories; key relevant finding: **synthetic tasks like NIAH do not reliably predict downstream performance**, and category scores correlate poorly with each other. + +**โญ Memora: From Recall to Forgetting** โ€” Md Nayem Uddin, Kumar Shubham, Eduardo Blanco, Chitta Baral, Gengyu Wang. arXiv:2604.20006, **21 Apr 2026**. Weeks-to-months conversations; introduces **FAMA (Forgetting-Aware Memory Accuracy)**, penalizing reliance on obsolete or invalidated memory. **Evaluating 4 LLMs and 6 memory agents: "frequent reuse of invalid memories and failures to reconcile evolving memories. Memory agents offer marginal improvements."** The strongest 2026 negative result in this space. + +Also *PerLTQA* (arXiv:2402.16288, 8,593 questions / 30 characters, semantic + episodic memory). + +### 7.3 Systems + +| System | Paper | Venue | Numbers | +|---|---|---|---| +| **MemGPT / Letta** | Packer et al., arXiv:2310.08560 | preprint | OS-style paged memory; established DMR as its eval | +| **Zep / Graphiti** | Preston Rasmussen, Pavlo Paliychuk, Travis Beauvais, Jack Ryan, Daniel Chalef. arXiv:2501.13956, 20 Jan 2025 | preprint, no venue comment | **DMR 94.8% vs MemGPT 93.4%**; LongMemEval **up to +18.5%**, **โˆ’90% latency** | +| **Mem0** | Prateek Chhikara, Dev Khant, Saket Aryan, Taranjeet Singh, Deshraj Yadav. arXiv:2504.19413, 28 Apr 2025 | โš ๏ธ arXiv shows no venue; secondary sources say ECAI 2025 (unverified) | **+26% relative LLM-as-Judge over OpenAI memory**; graph variant +~2%; **โˆ’91% p95 latency**, **>90% token cost saving** vs full-context | +| **MemoRAG** | arXiv:2409.05591 | โ€” | Light long-range model builds global memory + draft clues; expensive model answers | +| **Cognitive Workspace** | arXiv:2508.13171 | preprint | **58.6% avg memory reuse vs 0% for traditional RAG**; 17โ€“18% net efficiency gain despite 3.3ร— operations; p<0.001, Cohen's d>23. โš ๏ธ **Effect sizes that large are a red flag for a self-defined metric, and "0% for RAG" is true by construction** | +| **Memory-R1** | arXiv:2508.19828 | preprint | RL (PPO/GRPO) over ADD/UPDATE/DELETE/NOOP memory ops. On LLaMA-3.1-8B: **+48% F1, +69% BLEU-1, +37% LLM-as-Judge**, trained on only **152 QA pairs**. Beats Mem0, LangMem, A-MEM | + +### 7.4 โš ๏ธ CONTESTED โ€” the Mem0 vs Zep benchmark dispute + +A real, documented, unresolved public dispute. **Provenance warning: the underlying claims are in arXiv papers; the rebuttals live in blog posts and a GitHub issue, not in peer review.** + +1. **Zep (arXiv:2501.13956, Jan 2025)** claims DMR 94.8% vs MemGPT 93.4% and LongMemEval +18.5% / โˆ’90% latency; separately publicized **~84% on LoCoMo**. +2. **Mem0 (arXiv:2504.19413, Apr 2025)** benchmarks 10 approaches on LoCoMo: **Mem0 at 67.13% LLM-as-Judge, +26% relative over OpenAI memory, and Zep at 65.99%.** +3. **Mem0 โ†’ Zep, GitHub issue getzep/zep-papers#5, filed 8 May 2025 by Deshraj Yadav (Mem0 CTO):** alleges Zep's 84% is invalid because Zep **included questions from LoCoMo's excluded adversarial 5th category in the numerator while excluding them from the denominator**; also alleges Zep modified the system prompt with timestamp-favoring instructions not given to baselines, changed the retrieval template vs its own prior DMR work, and reported a **single run** rather than Mem0's 10-run-with-variance standard. **Mem0's re-run of Zep: 58.44% ยฑ 0.20 โ€” a 25.56-point reduction.** โš ๏ธ As of the fetch, the issue was open with no Zep response in-thread. +4. **Zep โ†’ Mem0, blog "Lies, Damn Lies, and Statistics":** alleges Mem0 misconfigured Zep three ways โ€” assigned the **user role to both participants** in a graph designed for single user-assistant interaction; passed **timestamps appended to message text instead of Zep's dedicated `created_at` field**, breaking temporal reasoning; ran **searches sequentially rather than in parallel**, inflating latency. Zep's corrected numbers: **75.14% ยฑ 0.17** (vs Mem0's reported 65.99%), p95 search latency **0.632s** vs Mem0's reported 0.778s. +5. **Zep also attacks the benchmark:** LoCoMo conversations are only **16kโ€“26k tokens**, contain **no knowledge-update tests**, and have data-quality defects โ€” missing ground truths, multimodal errors, incorrect speaker attribution, ambiguous questions. + +**Net state of the Zep-on-LoCoMo number: 84% (Zep original) โ†’ 65.99% (Mem0's measurement) โ†’ 75.14% (Zep's correction) โ†’ 58.44% (Mem0's re-run of Zep's own pipeline). A ~26-point spread on the same system and same benchmark, with no neutral adjudication.** + +**Honest reading: both parties are vendors self-reporting on a 2024 academic benchmark that both agree is partly broken. No corrected figure has been independently replicated in a refereed venue. Treat all LoCoMo numbers from memory vendors as unreliable.** The academic side has moved on โ€” MemoryAgentBench, LongMemEval-V2, and Memora were all built partly because LoCoMo is inadequate. + +**โญ Independent counter-evidence.** "Beyond the Context Window: A Cost-Performance Analysis of Fact-Based Memory vs. Long-Context LLMs for Persistent Agents" โ€” Natchanon Pollertlam, Witchayut Kornsuwannawit. arXiv:2603.04814, **5 Mar 2026**. Mem0-based fact memory vs long-context GPT-5-mini on LongMemEval, LoCoMo, PersonaMemv2. +- **Long-context GPT-5-mini achieves *higher* factual recall on both LongMemEval and LoCoMo.** Memory is only competitive on PersonaMemv2. +- The real finding is the cost model: long-context cost grows per-turn even under prompt caching; memory read cost is roughly fixed after a one-time write. **At 100k context, memory becomes cheaper after ~10 turns**, break-even falling as context grows. +- **Interpretation: as of 2026, memory systems are an economic argument, not an accuracy argument** โ€” directly contradicting Mem0's paper framing. + +โš ๏ธ **Two false attributions circulating in blog aggregators, corrected:** (a) "Letta scored 49.0% on LongMemEval in independent evaluation (arXiv 2603.04814)" โ€” **that paper does not evaluate Letta or Zep at all**, only Mem0 vs long-context GPT-5-mini. Do not use it. (b) "Mem0's new algorithm hits 93.4% on LongMemEval vs Zep 63.8%" traces to **Mem0's own 2026 marketing blog**, not a paper. + +### 7.5 Query rewriting vs end-to-end retrieval in multi-turn + +The field splits into **conversational query rewriting (CQR)** โ€” produce a standalone de-contextualized query โ€” and **conversational dense retrieval (CDR)** โ€” encode the whole session end-to-end. + +**TREC iKAT** 2024 and 2025. iKAT 2025 has a **SIGIR 2026 resource paper** (ACM DOI 10.1145/3805712.3808591) and a NIST track overview; the 2025 edition moved to a **live API where systems must rewrite, retrieve, and ground in real time per turn**. โš ๏ธ **Methodological caveat from the track itself: manual runs use the human rewrite as input, which advantages them and leaves assessment less complete for automatic reformulations โ€” so CQR-vs-ceiling comparisons are biased.** Participant papers: RALI@TREC iKAT 2024 (arXiv:2412.07998), CFDA & CLIP @ TREC iKAT 2025 (arXiv:2509.15588), Adaptive Personalized Conversational IR (arXiv:2508.08634). + +**โญ The clearest 2026 evidence comes from SemEval-2026 Task 8 (MTRAGEval), built on IBM's MTRAG.** +- **Sifei @ SemEval-2026 Task 8** (arXiv:2606.28352): training-free hybrid dense+sparse retrieval with **controlled query rewriting** + cross-encoder reranking โ†’ **0.5453 nDCG@5, 3rd of 38 teams**, vs strongest baseline 0.4795. +- **The most useful negative finding: controlled conversational rewriting combined with last-turn concatenation gives consistent gains across domains, while retrieval-oriented rewrites โ€” keyword lists, hypothetical-document expansion โ€” consistently HURT**, by distorting intent and over-amplifying rare terms. +- Others: uva-irlab-conv (arXiv:2606.11945 โ€” learned sparse + listwise reranking, five complementary LLM reformulations fused via variance-aware nested RRF); H-RAG (arXiv:2605.00631); AILS-NTUA (arXiv:2603.10524). Earlier: arXiv:2406.18960. + +**Verdict: as of 2026, CQR has not been displaced by end-to-end conversational dense retrieval โ€” every competitive MTRAGEval system rewrites. But the *type* of rewrite matters and the naive "make it retrieval-friendly" instinct is empirically wrong: decontextualize and preserve intent, keep the raw last turn, don't keyword-ify.** + +### 7.6 Honest 2026 verdict + +1. **The problem is real and large:** โˆ’39% singleโ†’multi-turn (200k+ sims), โˆ’30% on LongMemEval, MTRAG failure concentrated on later/non-standalone/unanswerable turns. +2. **The cause is not purely retrieval.** Laban et al.'s decomposition โ€” unreliability, not aptitude โ€” means a memory layer can only fix part of it. +3. **Vendor memory-system accuracy claims are not trustworthy right now.** A 26-point swing on one system on one benchmark with no neutral referee, and both sides agree the benchmark is inadequate. +4. **Independent 2026 work is unkind to memory systems.** Memora: 6 agents, "marginal improvements," frequent reuse of invalidated memory. LongMemEval-V2: RAG-based memory 48.5% vs a plain coding agent at 69.3%. arXiv:2603.04814: long-context GPT-5-mini beats Mem0 on factual recall on both LongMemEval and LoCoMo. +5. **The surviving case for memory is economic, not qualitative:** fixed per-turn read cost vs context cost growing even under caching; break-even ~10 turns at 100k context. +6. **The unsolved competency is not recall โ€” it's update and forgetting.** ~60% on single-hop conflict resolution with GPT-4o; Memora's whole FAMA metric exists because systems keep citing superseded facts; LongMemEval flagged knowledge-updates and abstention in Oct 2024 and they remain the failure modes in 2026. +7. **Most promising direction with real numbers: learned memory management.** Memory-R1 (+48% F1 / +69% BLEU-1 / +37% judge on LLaMA-3.1-8B with 152 training pairs) suggests the ADD/UPDATE/DELETE policy is learnable and is where the headroom is โ€” consistent with (6). + +--- + +## 8. CROSS-CUTTING OBSERVATIONS AND THE FULL LIST OF OPEN DEBATES + +### 8.1 The recurring arc + +Every one of the seven areas shows the same shape. **A 2024 preprint from a named lab makes a large claim on an LLM-as-judge or vendor-internal evaluation** (Edge et al.'s 72โ€“83% GraphRAG win rates; Anthropic's 67% failure reduction; Jina's late chunking; Zep/Mem0's LoCoMo scores; Search-R1's +41%). **A successor wave builds on it.** Then **2025โ€“2026 controlled, same-backbone, full-test-set replications with standardized budgets shrink the effect to a few points, relocate it to a narrow query class, and reveal the cost was the real story.** + +**In every area, the strongest 2026 papers are benchmark, reproducibility, and cost papers โ€” not method papers.** BrowseComp-Plus (ACL 2026), Lighting the Way for BRIGHT (SIGIR 2026 Repro), the PLAID reproduction (SIGIR 2024 Repro), Fixed RAG Compression Collapses Measured Reader Scaling, Do LLM Attribution Metrics Transfer, RAG vs GraphRAG, BM25 Wins at Scale, Memora, and the MTEB RTEB governance decision. If you read only ten things from this report, read those. + +### 8.2 A second recurring pattern: the "obvious baseline was omitted" + +This charge appears independently in at least six places and is the most reliable predictor of a deflated result: +- PLAID papers omitted BM25 + ColBERT reranking (MacAvaney & Tonellotto). +- BRIGHT baselines silently used an undocumented BM25 variant (Sharifymoghaddam, Ge, Lin). +- Long-context-beats-RAG papers used relevance-ordered, badly-chunked RAG arms (DOS RAG / OP-RAG). +- Self-correction papers leaked oracle stopping signals and under-resourced their baselines (Huang et al., ICLR 2024). +- Compression papers evaluated on weak readers (Panthi & Abdelfattah). +- Chunking papers baselined against fixed-*character* splitting rather than sentence-aware splitting. + +### 8.3 The complete open-debate list + +1. **Is RL necessary for agentic search, or is it trajectory-data quality?** LiteResearcher measures +15.7 GAIA from RL over its own SFT; OpenSeeker-v2 beats it on BrowseComp/HLE with **pure SFT** on 10.6k curated trajectories. No controlled experiment at matched scale exists. +2. **Do RL-trained search agents generalize?** BrowseComp-Plus: Search-R1-32B scores **exactly its untrained base model's score** on an out-of-distribution corpus. Unrebutted. +3. **Reasoning-aware retrieval: training problem or test-time-compute problem?** LATTICE (Google) matches the best fine-tuned ensembles with an off-the-shelf LLM. Never evaluated against the training camp at matched compute. +4. **Is the single-vector dimensional bound practically binding?** Weller et al. (DeepMind, ICLR'26) yes; Bangachev et al. (MIT) the bound is far looser; Spectral Retrieval lifts LIMIT-small R@10 0.33โ†’0.90 without retraining. +5. **Does late interaction earn its cost?** SIGIR'24 reproduction says BM25+ColBERT rerank at 9 ms/q beats the fastest PLAID at 73 ms/q. Not directly rebutted with an updated head-to-head. +6. **Are late interaction and learned sparse the same thing?** ColBERTSaR proves quantized ColBERT โ‰ก learned sparse; PLAID clusters align 1:1 with tokens. Emerging synthesis, not consensus. +7. **RRF vs convex combination.** Bruch et al. (TOIS 2023) showed CC wins; every 2025โ€“26 applied paper uses RRF anyway; nobody has re-tested. +8. **Does BM25 still beat dense?** Yes on LIMIT and financial tables and at low latency; no by ~15 points on aggregate suites and by ~14 points in agentic loops. Task-conditional. +9. **How much does hybrid add?** +9.17 nDCG (arXiv:2502.20245) vs +3.1pp (KohakuRAG). Unexplained, unreplicated. +10. **How much does chunking strategy matter?** Resolves once you notice the disagreeing papers use different worst-case baselines (fixed-character vs fixed-token). +11. **Does contextual retrieval work as advertised?** No independent reproduction of 35/49/67 exists. +12. **Are compression numbers real?** Reader-strength confound flips 31% of model rankings and hides 80% of a reader upgrade. +13. **LC vs RAG.** Reconciles as "most LC-wins papers used a weak RAG arm," but LC has its own length-alone degradation problem. +14. **Is "lost in the middle" still true?** Largely trained away for simple retrieval in frontier models; order-sensitivity for composition over many chunks persists. +15. **Does self-verification help?** Resolved along the intrinsic/external axis, but two live sub-disputes: whether external feedback is sufficient (Feedback Friction says no), and whether reported self-correction gains survive controlling for base capability (error-depth says no). +16. **Do internal states encode truthfulness or just recall?** Orgad et al. (ICLR 2025) vs arXiv:2510.09033. Both cannot be right in their strong forms. +17. **Is semantic entropy worth its cost?** No refutation, but three 2026 papers claim single-pass alternatives match or beat it. +18. **Are specialized hallucination detectors better than LLM-as-a-Judge?** TRIVIA+ says LLM-as-Judge is competitive. Every vendor detector paper says otherwise. Note who benefits from each answer. +19. **Do finer citations help or hurt?** Sentence-level requirements degrade attribution 16โ€“276%, and the penalty grows with model scale โ€” contradicting prevailing benchmark design. +20. **Did GraphRAG pay off?** Multi-hop win contested between +3 points and +27 EM; global-summarization claim not replicated; cost is 41โ€“57ร— indexing and ~377ร— inference; and at scale it cannot be built. +21. **Do memory systems beat long context?** Independent 2026 work says long context wins on recall; memory wins only on cost past ~10 turns. + +--- + +## 9. WHAT I COULD NOT VERIFY โ€” DO NOT CITE THESE AS FACTS + +**Numbers/claims:** +- Current (Aug 2026) MTEB/MMTEB #1. The leaderboard is a client-rendered Gradio Space; every fetch returned the loading shell. +- Numeric tables in arXiv:2605.05242 (grep-based Direct Corpus Interaction vs BM25/dense/rerankers). Existence, authorship (incl. Yejin Choi, Jiawei Han, Jimmy Lin), date, and qualitative claims verified; **the numbers are not.** Worth a follow-up given the author list. +- AttrScore per-model F1 and its human-agreement figure. +- ReDeEP's own reported AUROC (abstract omits it; 0.73 is third-party). +- TRIVIA+ detector AUROC table (PDF tables unreadable). +- Exact figures inside arXiv:2504.01818, arXiv:2604.03676, arXiv:2403.06789. +- 2026 numbers for NV-Embed, Stella, BGE-M3 successors, Voyage flagship tiers. +- Current Kaggle FACTS Grounding standings (page rendered empty). +- HHEM figures for GPT-5 Pro / Claude Opus 4.7 that appear in SEO summaries but **not** on the actual Vectara leaderboard. +- Lynx 2.0 numbers (blog-only). Galileo Hallucination Index methodology. +- "BEIR is no longer zero-shot because researchers train on it" and "MTEB has 400+ models with marginal differences" โ€” search-snippet only. +- SAAS, AutoSearch, and Agentic-R report no numbers in their abstracts. + +**Venues:** +- **A peer-reviewed venue for Edge et al. (GraphRAG, arXiv:2404.16130) โ€” likely never formally published.** +- ICLR 2026 for GraphRAG-Bench (2506.05690) and Youtu-GraphRAG (2508.19855) โ€” asserted in GitHub titles only. +- ECAI 2025 for Mem0 (2504.19413) โ€” secondary sources only. +- TACL for MTRAG โ€” IBM's page says so; arXiv has no comment. +- PathRAG (2502.14902). +- arXiv ID for Stechly/Valmeekam/Kambhampati ICLR 2025 (search said 2402.08115; unverified). +- "Gosling Grows Up" (SIGIR 2025) โ€” arXiv ID not verified and deliberately not guessed. + +**Corrections to things circulating as fact:** +- There is **no NVIDIA "RAG vs GraphRAG" study**; the systematic evaluation is arXiv:2502.11371 from **Michigan State + Meta + IBM**. +- There is **no academic paper titled "GraphRAG under fire."** +- **"Letta 49.0% on LongMemEval per arXiv:2603.04814" is false** โ€” that paper evaluates Mem0 vs long-context only. +- **arXiv:2601.04618 (REPAIR) was withdrawn by its authors on 14 Apr 2026.** Do not cite its +5.6pp. +- **Reason-ModernColBERT has no paper**, and its "beats ReasonIR-8B by 2.5 nDCG" claim holds only on the StackExchange splits; it loses on the full BRIGHT mean. +- **LettuceDetect is not SOTA on RAGTruth**; fine-tuned Llama-3-8B at 83.9% is. + +**Citation counts** (Semantic Scholar API, 8 Aug 2026; the API was rate-limited for most of this session and OpenAlex badly undercounts preprints, so this is a partial set): GraphRAG 1,875 ยท Search-R1 1,317 ยท LLMs Get Lost 378 ยท BRIGHT 177 (ICLR) ยท BrowseComp-Plus 155 ยท ReasonIR 76 ยท Rank1 72. GitHub traction as a secondary proxy: Alibaba-NLP/DeepResearch 19.8k stars ยท Search-R1 5.3k ยท FlashRAG 3.5k ยท PyLate 877 ยท BrowseComp-Plus 327 ยท ReasonIR 230. \ No newline at end of file diff --git a/Documentation/research/component-map-2026.md b/Documentation/research/component-map-2026.md new file mode 100644 index 00000000..2daf9771 --- /dev/null +++ b/Documentation/research/component-map-2026.md @@ -0,0 +1,1007 @@ +# STATE OF THE ART: LOCAL/SELF-HOSTED AGENTIC RETRIEVAL STACK +## Component-by-component map, as of 8 August 2026 + +**Method note:** Primary sources only โ€” arXiv, official GitHub repos, official HuggingFace model cards, official leaderboards (OmniDocBench, BRIGHT, MTEB/RTEB, LLM-AggreFact), and named vendor engineering blogs (Anthropic, Jina AI, Chroma, Weaviate, Qdrant, Naver Labs, Vectara, Zep, Letta, IBM, Microsoft Bing, Ai2, LlamaIndex). The session-wide WebSearch quota (200) was exhausted partway through; the remainder was gathered by direct fetch of primary URLs and the arXiv API. This biases coverage *toward* arXiv and *away* from vendor blogs โ€” a gap I flag rather than paper over. Confidence labels: **established** / **emerging** / **contested**. + +--- + +# 1. DOCUMENT PARSING / INGESTION + +## 1.1 The headline: sub-1B specialist VLMs have won the parsing benchmark, decisively + +**Claim: PaddleOCR-VL-1.6 (0.9B) leads OmniDocBench v1.6_full at 96.34 overall, ahead of MinerU2.5-Pro (1.2B, 95.75) and GLM-OCR (0.9B, 95.22). Every top slot is an open-weight specialist VLM; the best API model (Gemini 3 Pro) sits at 92.91 and GPT-5.2 at 86.59.** +โ†’ https://github.com/opendatalab/OmniDocBench (leaderboard fetched 2026-08-08) + +| Model | Type | Size | Overallโ†‘ | TextEditโ†“ | FormulaCDMโ†‘ | TableTEDSโ†‘ | +|---|---|---|---|---|---|---| +| PaddleOCR-VL-1.6 | specialist VLM | 0.9B | **96.34** | 0.0326 | 97.53 | 94.76 | +| MinerU2.5-Pro | specialist VLM | 1.2B | 95.75 | 0.036 | 97.45 | 93.42 | +| GLM-OCR | specialist VLM | 0.9B | 95.22 | 0.044 | 97.18 | 92.83 | +| PaddleOCR-VL-1.5 | specialist VLM | 0.9B | 94.93 | 0.038 | 96.89 | 91.67 | +| Youtu-Parsing | specialist VLM | โ€” | 93.74 | 0.044 | 93.63 | 92.02 | +| Gemini 3 Pro | API | โ€” | 92.91 | 0.064 | 95.99 | 89.15 | +| dots.ocr | specialist VLM | 3B | 90.77 | โ€” | โ€” | โ€” | +| GPT-5.2 | API | โ€” | 86.59 | 0.114 | 88.21 | 82.95 | + +**Confidence: established.** This is the single cleanest result in the entire survey. A 0.9B open-weight model that fits in ~2GB of VRAM beats every frontier API model at document parsing. + +**Claim: GLM-OCR (Zhipu/Z.ai, 0.9B) was released 2026-03-11 under MIT (layout component Apache 2.0), scoring 94.62 on OmniDocBench v1.5 โ€” the highest of any model, open or closed, at the time. Architecture: 0.4B CogViT encoder + connector + 0.5B GLM decoder. Reported throughput 1.86 PDF pages/sec via Multi-Token Prediction.** +โ†’ https://arxiv.org/pdf/2603.10910 (tech report, CC-BY 4.0) ยท https://www.llamaindex.ai/blog/omnidocbench-is-saturated-what-s-next-for-ocr-benchmarks (2026-02-24) +**Confidence: established** for the score and license; **emerging** for the throughput figure (vendor self-report). + +**Claim: PaddleOCR-VL-0.9B pairs a NaViT-style dynamic-resolution encoder with ERNIE-4.5-0.3B and supports 109 languages.** +โ†’ https://arxiv.org/abs/2510.14528 (v1 2025-10-16, v4 2025-11-25) ยท v1.6 tech report https://arxiv.org/pdf/2606.03264 +**Confidence: established.** + +**Claim: MinerU2.5 (1.2B) uses a two-stage coarse-to-fine strategy โ€” layout analysis on a downsampled image, then native-resolution crops โ€” and scores 75.2 on olmOCR-Bench (leading arXiv Math 76.6, Old Scans Math 54.6, Long Tiny Text 83.5).** +โ†’ https://arxiv.org/abs/2509.22186 (2025-09-26) +**Confidence: established.** The decoupled layoutโ†’crop pattern is now the dominant architecture; PaddleOCR-VL and GLM-OCR both use variants of it. + +## 1.2 The benchmark itself is saturating + +**Claim: OmniDocBench is saturated โ€” top models cluster at 94%+ on 1,355โ€“1,651 pages across 9โ€“10 document types, leaving "edge case fixing." LlamaIndex argues the field needs a benchmark rewarding semantic correctness over exact match, because agents care about functional accuracy, not formatting.** +โ†’ https://www.llamaindex.ai/blog/omnidocbench-is-saturated-what-s-next-for-ocr-benchmarks (2026-02-24) +**Confidence: emerging** (vendor opinion piece) but corroborated by the leaderboard's own compression at the top. + +**Claim: OmniDocBench maintainers responded with methodology changes rather than a new benchmark. Timeline: 2025-09-25 v1.0โ†’v1.5 (hybrid matching); 2026-04-10 v1.5โ†’v1.6 introducing Multi-Granularity Adaptive Matching (MGAM) to eliminate matching bias, +296 new pages; 2026-04-30 v1.7 with skills-based evaluation; 2026-07-27 community EvalScope integration for OpenAI-compatible endpoints.** +โ†’ https://github.com/opendatalab/OmniDocBench +**Confidence: established.** + +## 1.3 The olmOCR lineage has stalled + +**Claim: olmOCR 2 (olmOCR-2-7B-1025, Ai2, 2025-10-22) introduced RLVR with binary unit tests as rewards, scoring 82.4 on olmOCR-Bench (+~4 over the prior release), beating Marker (76.1) and MinerU (75.8). olmOCR-Bench runs 8,413 unit tests over 1,403 PDF pages. Permissive open licenses for model, data, and code.** +โ†’ https://arxiv.org/abs/2510.19817 ยท https://allenai.org/blog/olmocr-2 +**Confidence: established.** + +**Claim: No new olmOCR model shipped between Oct 2025 and Aug 2026. GitHub releases through v0.4.27 (2026-03-12) are pipeline/infrastructure work (PII tagging, timeouts, GPU deps) โ€” no model version bump and no benchmark movement.** +โ†’ https://github.com/allenai/olmocr/releases (fetched 2026-08-08) +**Confidence: established.** This is a real finding: **the Western open-OCR lineage went quiet for ~10 months while Chinese labs (Baidu/PaddleOCR, Zhipu/GLM, OpenDataLab/MinerU) shipped three generations.** olmOCR's lasting contribution is now olmOCR-Bench (independently maintained, not tied to the models it evaluates) rather than the models. + +## 1.4 Docling vs unstructured โ€” the framing changed + +**Claim: Docling is governed under the Linux Foundation AI & Data Foundation, MIT-licensed, and in 2026 became an orchestration layer that plugs in whichever VLM is currently best, rather than competing on parsing quality itself.** 2026 changelog additions: DeepSeek-OCR (v2.68), **GLM-OCR with vLLM backend (v2.84)**, LightOnOCR-2-1B and Falcon-OCR (v2.85), Nanonets-OCR2 (v2.87), Nemotron-OCR (v2.105); plus threaded docling-parse v6 backend (v2.96), `docling-slim` modularization (v2.92), pluggable VLM runtime with presets (v2.73), VideoPipeline + ASR (v2.108โ€“v2.116), chunking options in the service datamodel (v2.117). +โ†’ https://github.com/docling-project/docling ยท https://raw.githubusercontent.com/docling-project/docling/main/CHANGELOG.md (through v2.118.1, 2026-08-07) ยท original tech report https://arxiv.org/abs/2501.17887 (2025-01-27, AAAI 2025 workshop) +**Confidence: established.** + +**Claim: Docling supports both local VLM inference (Transformers + MLX) and remote OpenAI-compatible endpoints (vLLM, Ollama). Measured on a MacBook M3 Max: SmolDocling/MLX 6.15 s/page, Qwen2.5-VL-3B/MLX 23.50 s, Granite Vision/Transformers 104.75 s, Pixtral-12B/Transformers 1,828.21 s.** +โ†’ https://docling-project.github.io/docling/usage/vision_models/ +**Confidence: established.** The ~300ร— spread between MLX-small and Transformers-large is the operative fact for laptop-class local deployment. + +**Claim: Granite-Docling-258M (IBM, 2025-09-17, Apache 2.0) emits DocTags โ€” a unified markup for layout + text + semantics โ€” and substantially beats SmolDocling-256M: Table TEDS-with-content 0.96 vs 0.76, code recognition F1 0.988 vs 0.915, full-page OCR F1 0.84 vs 0.80, equations 0.968 vs 0.947. IBM explicitly states it "is designed to complement the Docling library, not replace it" and is "not intended for general image understanding."** +โ†’ https://huggingface.co/ibm-granite/granite-docling-258M +**Confidence: established.** + +**Claim: unstructured's differentiator is typed semantic elements (Title, NarrativeText, Table, ListItem, Header) rather than flat Markdown โ€” useful when downstream chunking logic keys off element type.** +โ†’ Synthesized from parser-comparison coverage; **I could not verify this against an official unstructured source within budget.** +**Confidence: emerging / partially unverified.** Treat the typed-element claim as accurate (it is unstructured's documented core abstraction) but the competitive positioning as unsourced. + +## 1.5 What's winning for local pipelines in 2026, and why + +**The 2026 local pipeline is: Docling as orchestration + a 0.9โ€“1.2B specialist VLM as the parsing engine.** Reasons, each grounded above: +1. **Quality is settled** โ€” a 0.9B specialist beats GPT-5.2 by ~10 points on OmniDocBench (96.34 vs 86.59), so there is no quality argument for an API. +2. **The models fit** โ€” 0.9Bโ€“1.2B at 4-bit runs on a laptop; MLX paths give 6โ€“24 s/page on Apple silicon. +3. **Licensing is genuinely permissive** โ€” GLM-OCR MIT, Granite-Docling Apache 2.0, MinerU2.5 open-weight, Docling MIT. +4. **Docling absorbed the churn** โ€” it added five new OCR backends in 2026, so the orchestration layer is stable while the engine underneath is swappable. You are not betting on one model. +5. **Traditional pipeline parsers survive as the fast path, not the quality path** โ€” Docling's own docs still route clean digital PDFs through docling-parse (now threaded, v6) and reserve the VLM for scanned/complex pages. + +**Caveat:** the benchmark is saturated and does not reward semantic correctness. If your documents are financial presentations or legal filings, the leaderboard gap between the top five models is inside the noise floor of your actual task. + +--- + +# 2. CHUNKING + +## 2.1 The evidence base โ€” and it points at boring defaults + +**Claim: In the Chroma technical report, plain RecursiveCharacterTextSplitter at 400 tokens with no overlap scored 88.1โ€“89.5% recall; ClusterSemanticChunker reached 0.913 and LLMSemanticChunker 0.919. Total spread across all methods was ~9% recall. Recommendation: recursive at 200โ€“400 tokens no overlap for simplicity, ClusterSemanticChunker if complexity is acceptable.** +โ†’ https://www.trychroma.com/research/evaluating-chunking (2024-07-03) ยท https://github.com/brandonstarxel/chunking_evaluation +**Confidence: established**, but **a 2024 result** โ€” it predates long-context embedders and late chunking maturity. + +**Claim (the important 2026 update): an eight-method / nine-dataset evaluation found advanced chunking introduces "substantially higher computational overhead" without meaningful effectiveness gains. Recursive Semantic 89.36 and Fixed-Size 87.71 Accuracy@5 lead; DenseX trails at 69.10. Fixed-size ran in <1 second where DenseX averaged 15+ hours; several methods hit 48-hour timeouts or OOM. LumberChunker scored highest on LLM-judged answer generation (4.35) but "results were unreliable due to frequent failures."** +โ†’ https://arxiv.org/html/2606.00881v1 (2026-05-30). Methods: Fixed-Size, Recursive Semantic, Sequential HAC, TextTiling, Max-Min, GraphSeg, LumberChunker, DenseX. Datasets: GutenQA, LiteraryQA, NovelQA, Qasper, TriviaQA, SQuAD, PoQuAD, Natural Questions. +**Confidence: emerging** (single preprint) but it is the largest controlled chunking comparison to date and it **replicates Chroma's 2024 conclusion two years later with a bigger grid.** + +## 2.2 Late chunking โ€” real, but narrower than the hype + +**Claim: Late chunking embeds the full document with a long-context embedder, then pools token embeddings into per-chunk vectors after the transformer. No additional training required; applies to any long-context embedder.** +โ†’ https://arxiv.org/abs/2409.04701 (v1 2024-09-07, v3 2025-07-07, Jina AI) +**Confidence: established** as a method. + +**Claim (contested): head-to-head evaluations find late chunking is *efficient* but *loses on relevance* to contextual retrieval. One study: late chunking "offers higher efficiency but tends to sacrifice relevance and completeness," while contextual retrieval "preserves semantic coherence more effectively but requires greater computational resources." A May 2026 comparison on NFCorpus with jina-v3 found ContextualRankFusion better overall than late chunking, with efficacy varying by dataset and model.** +โ†’ https://arxiv.org/abs/2504.19754 (2025-04-28, ECIR 2025 workshop) ยท corroborating 2026 comparison surfaced but not independently fetched +**Confidence: contested.** Late chunking's sign flips by corpus. It is a *cheap* fix for cross-reference breakage, not a general quality upgrade. + +## 2.3 Contextual retrieval โ€” still the strongest measured chunking intervention + +**Claim: Anthropic's contextual retrieval (prepend an LLM-generated ~100-token context blurb to each chunk at index time) reduces top-20 retrieval failure rate by 35% (contextual embeddings alone, 5.7%โ†’3.7%), 49% (+contextual BM25, โ†’2.9%), and 67% (+reranking, โ†’1.9%). Cost โ‰ˆ $1.02 per million document tokens with prompt caching, assuming 800-token chunks.** +โ†’ https://www.anthropic.com/news/contextual-retrieval (2024-09-19) +**Confidence: established.** Nearly two years old and still the reference number. Note the stacking: **most of the win comes from BM25 and reranking, not from the contextual blurb alone.** + +## 2.4 "No chunking, long-context embedders" + +**Claim: long context generally beats RAG on Wikipedia-style QA; summarization-based retrieval is comparable; chunk-based retrieval lags. But open-source LLMs "exhibit limited capacity for processing long contexts and therefore benefit substantially from retrieval," while closed models with stronger long-context ability do better on full context. Practical guidance from the same line of work: retrieval units should be longer and the number of chunks low โ€” top-5 to top-10 typically suffices.** +โ†’ https://arxiv.org/abs/2501.01880 (2025-01-03) +**Confidence: established for the asymmetry.** **This is the load-bearing fact for a self-hosted stack: the "just use long context" argument is an argument about frontier models, and it does not transfer to a local 7Bโ€“32B.** + +**Claim: even for strong models, retrieval helps on reasoning benchmarks โ€” a minimal RAG pipeline over CompactDS gave +10% MMLU, +33% MMLU Pro, +14% GPQA, +19% MATH relative, holding across 8Bโ€“70B.** +โ†’ https://arxiv.org/abs/2507.01297 (2025-07-02) +**Confidence: established.** + +**Cross-reference (see ยง10):** compression benefit shrinks as the reader gets stronger โ€” significant in 9/10 settings โ€” and a compressor tuned for a weak reader hides the gain from upgrading it (https://arxiv.org/abs/2606.21807, 2026-06-20). The same logic applies to chunking: **re-run your chunking ablation when you upgrade your local model.** + +## 2.5 Defaults people ship + +| Strategy | Ship it? | Evidence | +|---|---|---| +| Recursive/fixed 400โ€“600 tokens, minimal overlap | **YES โ€” the default** | 88โ€“89% recall (Chroma 2024); 87.7โ€“89.4 Acc@5 and best cost profile (2026-05-30) | +| Structure-aware (respect headings/tables from the parser) | **YES where the parser gives you structure** | Granite-Docling DocTags / unstructured typed elements make this nearly free | +| Contextual retrieval (index-time enrichment) | **YES if you can afford the offline pass** | 35โ€“67% failure-rate reduction (Anthropic 2024-09-19) | +| Late chunking | **Conditional** โ€” use when cross-references break | Efficient, but loses relevance vs contextual retrieval (2025-04, 2026-05) | +| Semantic / cluster chunking | **Only after measuring a retrieval gap** | +2โ€“3 pts recall at meaningfully higher cost | +| LLM-driven chunking (LumberChunker, DenseX) | **NO** | DenseX 69.1 Acc@5, 15+ hrs; LumberChunker "frequent failures" | +| No chunking / long context only | **NO for local models** | Open-source LLMs "benefit substantially from retrieval" (2025-01) | + +**The two-year-stable conclusion: chunking is not where your quality is. Every controlled study since 2024 puts the total spread across all chunking methods at ~9 points of recall, while a reranker alone is worth ~17 points of MRR@3 (see ยง5). Spend there instead.** + +--- + +# 3. EMBEDDINGS (LOCAL) + +## 3.1 Correction of a widespread error + +Multiple secondary sources date Qwen3-Embedding to "early 2026." **It is June 2025.** Verified: https://huggingface.co/Qwen/Qwen3-Embedding-8B and https://arxiv.org/abs/2506.05176 (submitted 2025-06-05, v3 2025-06-11), Apache 2.0. + +## 3.2 The scale ladder as of Aug 2026 (MMTEB / MTEB v2 multilingual mean) + +| Model | Params | Dims | Ctx | MMTEB v2 | License | Date | Source | +|---|---|---|---|---|---|---|---| +| **harrier-oss-v1-27b** (Microsoft Bing) | 27B | 5,376 | 32K | **74.3** | **MIT** | 2026-03/04 | [HF](https://huggingface.co/microsoft/harrier-oss-v1-27b) | +| **KaLM-Embedding-Gemma3-12B-2511** (Tencent) | 11.76B | 3,840 | 32K | **72.32** | Tencent-KaLM-Community | 2025-11 | [HF](https://huggingface.co/tencent/KaLM-Embedding-Gemma3-12B-2511) | +| **Qwen3-Embedding-8B** | 8B | 4,096 | 32K | 70.58 | **Apache 2.0** | 2025-06-05 | [HF](https://huggingface.co/Qwen/Qwen3-Embedding-8B) | +| Qwen3-Embedding-4B | 4B | 2,560 | 32K | 69.45 | Apache 2.0 | 2025-06-05 | same | +| **harrier-oss-v1-0.6b** | 0.6B | 1,024 | 32K | **69.0** | **MIT** | 2026-03/04 | [HF](https://huggingface.co/microsoft/harrier-oss-v1-0.6b) | +| **jina-embeddings-v5-text-small** | 677M | 1,024 | 32K | 67.0โ€“67.7 | CC BY-NC 4.0 | 2026-02-18 | [HF](https://huggingface.co/jinaai/jina-embeddings-v5-text-small) | +| **harrier-oss-v1-270m** | 270M | 640 | 32K | **66.5** | **MIT** | 2026-03/04 | [HF](https://huggingface.co/microsoft/harrier-oss-v1-270m) | +| jina-embeddings-v5-text-nano | 239M | 768 | 8K | 65.5 | CC BY-NC 4.0 | 2026-02-18 | [Jina blog](https://jina.ai/news/jina-embeddings-v5-text-distilling-4b-quality-into-sub-1b-multilingual-embeddings/) | +| Qwen3-Embedding-0.6B | 0.6B | 1,024 | 32K | 64.33 | Apache 2.0 | 2025-06-05 | HF | +| **EmbeddingGemma-300m** | 300M | 768 | **2K only** | 61.15 | Gemma terms | 2025-09 | [model card](https://ai.google.dev/gemma/docs/embeddinggemma/model_card) | + +English MTEB v2 where reported: Qwen3-Embedding-8B **75.22**, 4B 74.60; jina-v5-text-small **71.7**, nano 71.0; EmbeddingGemma 69.67. +**Confidence: established** (all from official model cards / vendor blogs). Note vendor self-report bias is inherent; the MMTEB leaderboard itself renders client-side and could not be fetched directly this session. + +## 3.3 Practical picks by scale โ€” and license is the deciding factor + +- **~270โ€“300M (edge/CPU):** **harrier-oss-v1-270m (MIT, 66.5, 32K ctx)** now dominates EmbeddingGemma-300m (61.15, **2K ctx**) on both quality and context length. EmbeddingGemma's remaining advantage is QAT checkpoints (int4/int8, Q8_0, Q4_0) with minimal degradation and a mature on-device story. +- **~0.6B (the sweet spot):** **harrier-oss-v1-0.6b (MIT, 69.0)** > jina-v5-text-small (67.0, **non-commercial**) > Qwen3-Embedding-0.6B (64.33, Apache 2.0). If you need a permissive license and 32K context, harrier-0.6b is the 2026 default and it is *4.7 points* above Qwen3-0.6B. +- **4Bโ€“8B:** **Qwen3-Embedding-4B/8B (Apache 2.0)** remains the safe permissive choice, and it is the retriever that moved BrowseComp-Plus by +14.2 points when swapped in for BM25 (ยง8). +- **12Bโ€“27B:** harrier-oss-v1-27b (MIT, 74.3) if you have the VRAM; KaLM-Gemma3-12B (72.32) is close but carries a bespoke Tencent community license. + +**Licensing summary โ€” this is the most practically consequential fact in this section:** MIT/Apache options are **harrier-oss-v1 (all three, MIT)** and **Qwen3-Embedding (all three, Apache 2.0)**. **jina-embeddings-v5 is CC BY-NC 4.0** and **KaLM is a bespoke community license** โ€” neither is usable in a commercial self-hosted product without contacting the vendor. + +## 3.4 Matryoshka truncation โ€” measured degradation is small and roughly linear + +**Claim: EmbeddingGemma MRL degradation, 768d โ†’ 128d (6ร— storage reduction): multilingual MTEB v2 61.15 โ†’ 58.23 (โˆ’2.92); English v2 69.67 โ†’ 66.66 (โˆ’3.01). Intermediate: 512d costs ~0.44 pts, 256d ~1.4 pts.** +โ†’ https://ai.google.dev/gemma/docs/embeddinggemma/model_card +**Confidence: established.** This is the best publicly documented MRL degradation curve. + +**Claim: MRL is now universal at the top of the leaderboard.** Qwen3-Embedding supports 32โ€“4096; jina-v5-text-small supports 32/64/128/256/512/768/1024; KaLM-12B supports 3840/2048/1024/512/256/128/64; Qwen3-VL-Embedding supports MRL. +โ†’ respective model cards +**Confidence: established.** + +**Claim: Jina's GOR regularization makes binary quantization "nearly lossless," dropping the effective memory footprint of a production embedding service "by an order of magnitude."** +โ†’ https://jina.ai/news/jina-embeddings-v5-text-distilling-4b-quality-into-sub-1b-multilingual-embeddings/ (2026-02-19) +**Confidence: emerging** (vendor claim, no independent replication). If it holds, binary + MRL together is a ~50ร— storage reduction, which matters more for local deployment than 2 points of MMTEB. + +**Practical rule:** truncate to **256d as the default** (~1.4 pt cost, 3ร— storage saving); go to 128d only if storage-bound; do not go below 128d without measuring. + +## 3.5 Instruction-tuned embedders are now the norm, not a variant + +**Claim: every top-tier 2026 embedder is query-instruction-aware, and all use the same `Instruct: {task}\nQuery: {text}` format with no instruction on the document side.** Qwen3-Embedding reports **1โ€“5% improvement** from task-specific instructions. Harrier's card states flatly: "Each query must come with a one-sentence instruction that describes the task." +โ†’ https://huggingface.co/Qwen/Qwen3-Embedding-8B ยท https://huggingface.co/microsoft/harrier-oss-v1-0.6b +**Confidence: established.** Practical consequence: **if you are running one of these models without an instruction prefix, you are leaving 1โ€“5% on the table and your indexed documents are fine โ€” only the query path changes.** + +**Claim: Jina v5 takes a different route โ€” four task-specific LoRA adapters (retrieval, text-matching, classification, clustering) on one base, distilled from Qwen3-Embedding-4B, with adapter-merged checkpoints published for vLLM and TEI.** +โ†’ https://huggingface.co/jinaai/jina-embeddings-v5-text-small +**Confidence: established.** Note the distillation source: **a 677M model distilled from Qwen3-Embedding-4B matches jina-v4 (3.8B) on retrieval at 5.6ร— smaller.** Distillation-from-a-bigger-embedder is the 2026 recipe for the sub-1B tier โ€” harrier's 270m/0.6b also use knowledge distillation from larger models. + +## 3.6 Benchmark integrity: MTEB overfitting is now officially acknowledged + +**Claim: MTEB's own maintainers launched RTEB (Retrieval Embedding Benchmark) on 2025-10-01 specifically because "when models are repeatedly evaluated against the same public datasets, a gap emerges between their reported scores and their actual performance on new, unseen data." RTEB pairs open datasets with private held-back datasets across 20 languages and enterprise domains (law, healthcare, finance, code), scored on nDCG@10. The openโ†”private score gap directly measures overfitting.** +โ†’ https://huggingface.co/blog/rteb (2025-10-01) +**Confidence: established.** + +**Claim: MMTEB (ICLR 2025) found "the best-performing publicly available model is multilingual-e5-large-instruct with only 560 million parameters" โ€” i.e. parameter count did not predict rank at the time. 500+ tasks, 250+ languages.** +โ†’ https://arxiv.org/abs/2502.13595 (2025-02-19, final 2025-11-13) +**Confidence: established**, and now partly superseded โ€” the 2026 leaderboard is size-ordered again at the top (27B > 12B > 8B). + +**Practical advice: treat MMTEB deltas under ~2 points as noise, and validate your top 2โ€“3 candidates on your own held-out set. The +6.2pp accuracy swing from a bare embedding-model swap measured in the memory literature (ยง11) is larger than most architectural choices you will make.** + +--- + +# 4. SPARSE LEG + FUSION + +## 4.1 BM25 / native FTS locally + +**Claim: bm25s (pure Python + NumPy) achieves 1,196 QPS on nfcorpus vs rank-bm25 at 224.66 and Elasticsearch at 45.84 โ€” ~26ร— faster than Elasticsearch. v0.2.0 added a numba backend for ~2ร— further speedup on larger datasets.** +โ†’ https://github.com/xhluca/bm25s +**Confidence: established** (self-reported, but the methodology is public and reproducible). **For a local stack this settles it: you do not need Elasticsearch for the sparse leg.** SQLite FTS5, DuckDB FTS, Postgres tsvector, Tantivy, and bm25s are all adequate; the choice is an operational one, not a quality one. + +**Claim: "BM25 remains the first-stage retrieval mechanism in many high-throughput production systems even when dense retrieval is layered on top" โ€” SPLADE outperforms BM25 on BEIR but requires GPU inference, where BM25 is a CPU-only inverted index.** +**Confidence: established** as the standard tradeoff, though I could not anchor this specific phrasing to a primary source within budget โ€” treat the *reasoning* as sound and the *quote* as unsourced. + +## 4.2 Learned sparse โ€” quietly excellent, and 2026 was a good year + +**Claim: SPLADE-v3 is "statistically significantly more effective than both BM25 and SPLADE++, while comparing well to cross-encoder re-rankers." It maps text to a 30,522-dim sparse vector (BERT vocab) with learned term weights and implicit expansion.** +โ†’ https://arxiv.org/abs/2403.06789 ยท https://github.com/naver/splade +**Confidence: established.** + +**Claim (the 2026 headline): SPLARE (Naver Labs Europe) is a new learned sparse retriever producing "generalizable sparse latent representations," available at 7B and a "significantly lighter" 2B. It achieves top results on MMTEB multilingual and English retrieval and beats state-of-the-art dense baselines including Qwen3-8B-Embed. Deployed at WSDM Cup 2026 with Qwen3-Reranker-4B and simple score fusion on top.** +โ†’ https://arxiv.org/abs/2602.20986 (2026-02-24) +**Confidence: emerging** (competition report; the underlying SPLARE paper's full results were not retrievable this session). **But "learned sparse beats Qwen3-8B-Embed on multilingual retrieval" is a significant claim that reopens the sparse-vs-dense question.** + +**Claim: SPLADE-Code (2026-03-23) is the first large-scale learned-sparse family for code retrieval, 600Mโ€“8B. Sub-1B variants hit 75.4 on MTEB Code (SOTA among sub-1B retrievers); the 8B reaches 79.0. Critically: "sub-millisecond retrieval on a 1M-passage collection with little effectiveness loss."** +โ†’ https://arxiv.org/abs/2603.22008 +**Confidence: emerging.** Directly relevant if your corpus is code. + +**Claim: a 2026 training fix โ€” rescaling the MLM-head projection by a constant at initialization โ€” resolves why stronger pretrained encoders (ModernBERT, Ettin) previously *failed* in SPLADE training due to inflated MLM-head L2 norms. It converts unstable runs into competitive sparse retrievers, in several cases matching or exceeding classic BERT-SPLADE.** +โ†’ https://arxiv.org/abs/2606.18811 (2026-06-17) +**Confidence: emerging.** This unblocks modern-encoder SPLADE, which had been stuck on BERT-base for years. + +**Practical read: learned sparse is no longer a research curiosity, but it costs you GPU inference at index time and query time. For most self-hosted stacks BM25 remains the right sparse leg; SPLADE/SPLARE is worth it if you are multilingual, code-heavy, or latency-critical (sub-ms at 1M passages).** + +## 4.3 Fusion โ€” RRF vs weighted vs learned + +**Claim: Weaviate's default fusion has been `relativeScoreFusion` (min-max normalize each list, add weighted normalized scores) since v1.24, not RRF. `rankedFusion` (RRF) "keeps only the position of a result in each list and discards the scores" and remains available via `fusionType`. Weaviate publishes no measured comparison between the two.** +โ†’ https://weaviate.io/blog/hybrid-search-explained +**Confidence: established.** Note the common claim that "Weaviate defaults to RRF" is **wrong** as of v1.24+. + +**Claim: Qdrant supports RRF (default k=2; weighted variant since v1.17.0) and DBSF (Distribution-Based Score Fusion, normalizing by mean/ฯƒ over a 3-sigma range). Its published decision table: tunable eval set โ†’ weighted RRF with train/val split; trust raw scores but no eval set โ†’ DBSF; no eval set or score priors โ†’ RRF ("the safe default"). Explicit caveats: score normalization relies on small top-k samples so single outliers skew it; "a fixed alpha over raw scores tends to be dominated by whichever retriever has larger raw magnitudes"; "neither method dominates universally"; retune when retrievers, embeddings or corpus change.** +โ†’ https://qdrant.tech/documentation/concepts/hybrid-queries/ +**Confidence: established.** This is the most honest vendor guidance I found โ€” it explicitly declines to claim a winner. + +**Claim: the theoretical case for RRF is that BM25 scores are unbounded positives while cosine is bounded [โˆ’1,1], so naive addition is meaningless; RRF sidesteps normalization entirely by using ranks with k=60 (or k=2 in Qdrant).** +**Confidence: established** as reasoning; the specific WANDS numbers circulating (RRF nDCG 0.7068 vs BM25 0.6983 vs KNN 0.6953) come from a secondary source I could not verify โ€” **do not cite those figures.** + +**Claim: hybrid+RRF measurably beats both legs. On T2-RAGBench (23,088 financial questions, 7,318 docs): Hybrid RRF Recall@5 0.695 / MRR@3 0.433, vs BM25 0.644/0.411 and dense 0.587/0.351. Adding a cross-encoder takes it to 0.816/0.605.** +โ†’ https://arxiv.org/html/2604.01733v1 (2026-04-02) +**Confidence: established** for this domain. Note **BM25 alone beat dense alone by 5.7 points of Recall@5** here โ€” a useful corrective for anyone considering dropping the sparse leg. + +**Claim (learned fusion): projection fusion outperforms RRF when combined with diversity reranking on TREC-COVID, comparing BM25/SPLADE sparse against BGE/DPR dense.** +โ†’ https://arxiv.org/pdf/2604.13728 (2026-04) +**Confidence: emerging / weakly supported** โ€” single-author preprint, single dataset, and the PDF's numeric tables were not extractable. **There is no strong 2026 evidence that learned fusion beats RRF in the general case.** + +## 4.4 Bottom line for ยง4 + +**Ship: BM25 (bm25s or your DB's native FTS) + a dense leg + RRF. Do not spend time tuning fusion before you have a reranker.** RRF is the safe default precisely because it is scale-free; weighted/DBSF only pays if you have an eval set to tune against, which is Qdrant's own published position. Learned sparse (SPLADE-v3 / SPLARE / SPLADE-Code) is a genuine 2026 upgrade path but costs GPU at index and query time. + +--- + +# 5. RERANKERS (LOCAL) + +## 5.1 The single highest-ROI component in the stack + +Across every 2026 head-to-head that includes both, reranking beats every query-side technique by roughly an order of magnitude in ROI: +- T2-RAGBench: hybrid 0.433 MRR@3 โ†’ **+cross-encoder 0.605** (**+17.2 pp**), Recall@5 0.695 โ†’ 0.816. โ†’ https://arxiv.org/html/2604.01733v1 (2026-04-02) +- Local 7B ablation: removing the reranker costs **โˆ’1.7 EM (p<0.001) at negligible latency**; the authors retain it "unconditionally." โ†’ https://arxiv.org/abs/2606.21553 (2026-06-19) +- Anthropic contextual retrieval: reranking took failure-rate reduction from 49% โ†’ **67%**. โ†’ https://www.anthropic.com/news/contextual-retrieval (2024-09-19) + +**Confidence: established.** This is the most replicated finding in the whole survey. + +## 5.2 Cross-encoders โ€” the bge lineage and Qwen3-Reranker + +**Claim: Qwen3-Reranker (0.6B/4B/8B, Apache 2.0, 2025-06-05) substantially beats bge-reranker-v2-m3 on the Qwen team's own evaluation:** + +| Model | Params | MTEB-R | CMTEB-R | MMTEB-R | MLDR | MTEB-Code | FollowIR | +|---|---|---|---|---|---|---|---| +| Qwen3-Reranker-0.6B | 0.6B | 65.80 | 71.31 | 66.36 | 67.28 | 73.42 | 5.41 | +| **Qwen3-Reranker-4B** | 4B | **69.76** | 75.94 | 72.74 | 69.97 | **81.20** | **14.84** | +| Qwen3-Reranker-8B | 8B | 69.02 | **77.45** | **72.94** | **70.19** | 81.22 | 8.05 | +| bge-reranker-v2-m3 | 0.6B | 57.03 | 72.16 | 58.36 | 59.51 | 41.38 | โˆ’0.01 | + +โ†’ https://qwenlm.github.io/blog/qwen3-embedding/ (2025-06-05) +**Confidence: established** (vendor self-report, but the FollowIR and MTEB-Code gaps are too large to be tuning artifacts). Two things stand out: **bge-reranker-v2-m3 is effectively useless on code (41.38) and on instruction-following (โˆ’0.01 FollowIR)**, and **4B beats 8B on several tasks** โ€” do not assume the 8B is the right pick. + +**Claim: bge-reranker-v2-m3 (0.6B, Apache 2.0, XLM-RoBERTa on bge-m3) has a 512-token max length. BAAI's own guidance: use it "for efficiency"; use bge-reranker-v2-minicpm-layerwise or -gemma "for maximum performance."** +โ†’ https://huggingface.co/BAAI/bge-reranker-v2-m3 +**Confidence: established.** **No bge-reranker-v3 exists as of Aug 2026** โ€” the newest in the family is bge-reranker-v2.5-gemma2-lightweight (July 2024). **The bge reranker lineage has been static for two years.** + +**Claim: mxbai-rerank-large-v2 (2B, Apache 2.0, 2025-06-04) scores BEIR 57.49 at 0.89s latency on A100; base-v2 (1.5B) 55.57 at 0.67s. Trained with RL + contrastive + preference learning.** +โ†’ https://huggingface.co/mixedbread-ai/mxbai-rerank-large-v2 +**Confidence: established.** Note the multilingual score is weak (29.79). + +## 5.3 Late-interaction and "not-late" interaction rerankers + +**Claim: jina-reranker-v3 (0.6B on Qwen3-0.6B, 2025-09-29) introduced "last but not late interaction" โ€” causal self-attention between query and all documents inside one shared 131K context window, up to 64 documents at once. BEIR 61.94 nDCG@10, SOTA, beating mxbai-rerank-large-v2 (61.44) with 2.5ร— fewer parameters, Qwen3-Reranker-0.6B (56.28) and bge-reranker-v2-m3 (56.51). Also HotpotQA 78.56, FEVER 93.95, CoIR 63.28. But MIRACL multilingual 66.50 โ€” *below* bge-reranker-v2-m3 (69.32) and Qwen3-Reranker-4B (67.52).** +โ†’ https://arxiv.org/html/2509.25085v1 ยท https://huggingface.co/jinaai/jina-reranker-v3 +**Confidence: established.** +**โš ๏ธ Critical licensing flag: jina-reranker-v3 is CC BY-NC 4.0 โ€” non-commercial.** The best-scoring small reranker on BEIR is not usable in a commercial self-hosted product without a Jina license. + +**Claim: KaLM-Reranker-V1 (Tencent, 2026-06-22, rev 2026-07-07) proposes "fast but not late interaction" (FBNL): an encoder pre-encodes passages with Matryoshka embedding pooling; a decoder processes instructions + query; cross-attention computes relevance. Sizes Nano 0.27B / Small 1B / Large 4B activated params. Claims SOTA on BEIR comparable to Qwen3-Reranker, strong MIRACL despite limited multilingual training, and "even the 0.27B Nano model remaining competitive with 7โ€“12B embedding models" on LMEB.** +โ†’ https://arxiv.org/abs/2606.22807 +**Confidence: emerging** โ€” no quantitative latency numbers in the abstract, PDF tables unextractable. Worth tracking as the likely permissive answer to jina-v3's NC license. + +**Claim: ColBERT-style late interaction persists as research (ColBERT-Att, 2026-03) but has produced no new production-grade open reranker in the last 12 months. PLAID remains the efficiency engine (up to 7ร— GPU / 45ร— CPU speedup over vanilla ColBERTv2).** +โ†’ https://arxiv.org/html/2603.25248v1 ยท https://arxiv.org/abs/2205.09707 +**Confidence: established.** **Late interaction lost the reranking race to "last-but-not-late" and cross-encoders in 2025โ€“26.** Its remaining stronghold is multimodal/document-image retrieval (ColPali lineage), not text reranking. + +## 5.4 LLM-as-reranker โ€” the numbers say no, with one exception + +**Claim (the cost/latency reality): pointwise, BGE-reranker-v2 scores 0.74 NDCG@10 at 12ms and $2/1k queries; Gemini Flash scores 0.68 at 185ms and $27/1k. Listwise, Gemini Flash reaches 0.78 at 420ms and $18/1k โ€” i.e. LLM listwise buys +0.04 NDCG@10 at ~35ร— the latency and ~9ร— the cost. Recommendation: specialized reranker narrows to top-20, LLM listwise only on the top-10.** +โ†’ https://zeroentropy.dev/articles/llm-as-reranker-guide/ (2025-07-20) +**Confidence: established** for the shape; the absolute numbers are one vendor's benchmark. The 35ร— latency multiplier is corroborated qualitatively across the literature. + +**Claim (2026 efficiency work is closing the gap): CompRank (Mistral-7B, 2026-06-10) decouples document representations via block-structured attention, compresses to 10.2% of document tokens, and uses attention-derived decoding-free scoring. BEIR nDCG@10 39.2 average (vs ICR-Mistral-7B 38.4, RankGPT-Mistral-7B 35.1, BM25 35.4), with 4.9ร—โ€“9.5ร— speedup over generation-based listwise reranking and 0.050 s/query query-block-forward at 500 documents.** +โ†’ https://arxiv.org/html/2606.11700v1 +**Confidence: emerging.** + +**Claim: whole-pool setwise reranking with long-context LLMs (DualEnd, 2026-06-01) ranks 100 candidates in 50 serial calls vs 99 for one-at-a-time whole-pool methods, evaluated across nine open-weight LLMs. Framing: "long context is not merely more prompt space, but an opportunity to make LLM re-rankers both effective and efficient."** +โ†’ https://arxiv.org/abs/2606.01782 +**Confidence: emerging.** + +## 5.5 Recommended reranker for a self-hosted stack + +| Constraint | Pick | Why | +|---|---|---| +| **Permissive license + best quality** | **Qwen3-Reranker-4B** (Apache 2.0) | MTEB-R 69.76, MTEB-Code 81.20, FollowIR 14.84 โ€” the only strong option that is both permissive and instruction-following | +| **Permissive + tight latency** | **Qwen3-Reranker-0.6B** (Apache 2.0) | 65.80 MTEB-R, +8.8 over bge-v2-m3, still sub-1B | +| **Best BEIR score, non-commercial OK** | jina-reranker-v3 (CC BY-NC) | BEIR 61.94, 131K ctx, 64 docs/pass | +| **Multilingual first** | bge-reranker-v2-m3 or Qwen3-Reranker-4B | bge still leads MIRACL (69.32); jina-v3 is weaker there (66.50) | +| **Apache + mid-size** | mxbai-rerank-large-v2 (2B) | BEIR 57.49, 0.89s A100; weak multilingual | +| **LLM reranker** | **Only as a top-10 second pass** | +0.04 NDCG for 35ร— latency / 9ร— cost | + +**Architectural note:** Qwen3's causal LM yes/no-logit design is fundamentally slower than a SequenceClassification single forward pass (bge/mxbai). Budget accordingly โ€” the quality gap is real but so is the throughput gap. + +**โš ๏ธ Cross-reference to ยง10:** rather than adding a compressor after your reranker, consider replacing the reranker with **Provence / OpenProvence**, which folds sentence-level context pruning *into* the reranking pass at ~zero marginal cost. + +--- + +# 6. QUERY PLANNING + +## 6.1 HyDE in 2026: demoted, not abandoned + +**Claim: HyDE consistently *underperforms* vanilla dense retrieval on entity-centric/numeric corpora. On T2-RAGBench (23,088 financial questions): HyDE Recall@5 0.544 / MRR@3 0.318 / nDCG@10 0.433 vs dense baseline 0.587 / 0.351 / 0.466 โ€” i.e. โˆ’7.3% / โˆ’9.4% / โˆ’7.1%. Stated cause: "LLM-generated pseudo-documents introduce noise through fabricated financial figures." CRAG/adaptive retrieval also underperformed plain hybrid fusion (0.658 vs 0.695 Recall@5).** +โ†’ https://arxiv.org/html/2604.01733v1 (2026-04-02) +**Confidence: established for numeric/entity domains.** + +**Claim (the most actionable HyDE result of the year): in a production RAG system, LLM augmentation was needed for only 27.8% of real queries, while synthetic eval sets implied >90% โ€” "the Coverage Illusion." Four ML approaches to *pre-retrieval* routing all failed: "the need for LLM augmentation cannot be determined from the query alone." The fix is a post-retrieval cascade โ€” escalate to HyDE only when initial retrieval returns nothing. Result vs always-HyDE: +0.140 Composite Overall, โˆ’31.8% latency, 72.2% of queries served with no LLM augmentation.** +โ†’ https://arxiv.org/abs/2605.27220 (Hussain & Nielbo, 2026-05-26) +**Confidence: emerging** (single production system, preprint) but structurally important. + +**Claim (counter-evidence): HyDE still wins in conversational multi-domain QA โ€” "robust yet straightforward methods, such as reranking, hybrid BM25, and HyDE, consistently outperform vanilla RAG," across 8 conversational QA datasets. And in Turkish, HyDE maximizes accuracy (85%) but the Pareto-optimal config reaches 84.60% at far lower cost.** +โ†’ https://arxiv.org/abs/2602.09552 (2026-02-10) ยท https://arxiv.org/abs/2602.03652 (2026-02-03) +**Confidence: contested โ€” HyDE's sign genuinely flips with corpus type.** + +**Claim: on small local models HyDE costs +25โ€“40% response time with high hallucination rates on personal queries.** +โ†’ https://arxiv.org/abs/2506.21568 (2025-06-12) โ€” โš ๏ธ *2025 result on Gemma 1B/4B, possibly stale.* + +**Claim: the 2026 architectural response is to move hypothetical generation to index time. HyPE precomputes hypothetical *questions* per chunk and matches questionโ†”question, reporting up to +42 pp context precision and +45 pp claim recall across six datasets, with zero query-time latency.** +โ†’ https://arxiv.org/abs/2607.29402 (arXiv 2026-07-31; โš ๏ธ *the underlying paper is IEEE Access 2025 โ€” treat as a 2025 result*) +**Confidence: emerging.** + +**Claim (docs lag): LlamaIndex still documents HyDE and multi-step decomposition with no caveats. LangChain's 2026 retrieval page leads with "2-Step RAG vs Agentic RAG vs Hybrid RAG" architecture selection and does not surface HyDE, multi-query, or self-query at all.** +โ†’ https://developers.llamaindex.ai/python/framework/optimizing/advanced_retrieval/query_transformations/ ยท https://docs.langchain.com/oss/python/langchain/retrieval (both fetched 2026-08-08) +**Confidence: established (direct observation).** LangChain quietly demoted query transformation in its 2026 information architecture; LlamaIndex did not. + +**Verdict: HyDE is alive but is no longer a default. Never always-on. Either gate it post-retrieval (cheap, no training) or move the hypothetical to index time.** + +## 6.2 Query decomposition โ€” real, modest, and stage-dependent + +**Claim (the best local-hardware ablation available): on Qwen2.5-7B-Instruct via Ollama on one RTX A6000, Qdrant + FastEmbed + BM25, HotpotQA distractor dev, n=5,000:** + +| Variant | EM | F1 | Latency (ms) | +|---|---|---|---| +| Baseline (single-pass dense) | 43.1 | 54.0 | **546** | +| Agentic full (3 steps) | 53.2 | 61.6 | 5,642 | +| no-decomposition | 51.9 | 60.1 | **2,546** | +| no-reranker | 51.5 | 59.4 | 5,648 | +| steps-1 | 46.1 | 53.9 | 5,426 | +| steps-2 | 52.9 | 61.3 | 5,628 | +| **hybrid-only (no adaptive routing)** | **55.0** | **63.5** | 5,688 | + +Key results: decomposition is worth **โˆ’1.4 EM when removed (p=0.004) but removing it halves latency**; **"two retrieval iterations capture 95% of the gain from five" (p<0.001)** with the 1โ†’2 jump worth +7.1 EM; reranking worth โˆ’1.7 EM at negligible latency. Verbatim recommendation: *"prefer fixed hybrid retrieval over rule-based routing, and cap retrieval depth at two or three steps."* Decomposition is *"the first component to drop under a tight latency budget."* +โ†’ https://arxiv.org/abs/2606.21553 (2026-06-19) โ€” โš ๏ธ *single-author preprint, unrefereed, but methodologically clean and the closest published match to a self-hosted stack.* +**Confidence: emerging.** + +**Claim (the most useful design rule of 2026): decomposition at initial retrieval "frequently harms performance due to semantic dilution," but "substantially improves reranking" via fine-grained constraint verification. Fix: keep the full query for first-stage retrieval, use sub-queries only at rerank.** +โ†’ https://arxiv.org/abs/2606.08577 (2026-06-07), benchmarks MultiConIR + SSRB, code at github.com/EIT-NLP/Query-Decompose +**Confidence: emerging.** + +**Claim: bandit-style adaptive decomposition (retrieve one doc at a time, update belief per sub-query) gives +35% document-level precision and +15% ฮฑ-nDCG.** +โ†’ https://arxiv.org/abs/2510.18633 (2025-10-21), **published EACL 2026** โ€” https://aclanthology.org/2026.eacl-long.322.pdf +**Confidence: established (peer-reviewed).** Note this is adaptive *budget allocation*, not naive fan-out. + +**Claim (conflicting): in a structured DevOps domain, agentic decomposition gave +0.04 overall / +0.17 MRR, but on MuSiQue multi-hop **ranking precision declined**. Conclusion: "agentic enhancements are not universally beneficial and must be applied selectively."** +โ†’ https://arxiv.org/abs/2606.05658 (2026-06-04) +**Confidence: emerging; conflicts in sign with the HotpotQA ablation โ€” genuinely contested.** + +## 6.3 Multi-query expansion โ€” the weakest of the three + +**Claim: on T2-RAGBench, multi-query scored Recall@5 0.640 / MRR@3 0.397 โ€” *worse than plain BM25* (0.644 / 0.411).** +โ†’ https://arxiv.org/html/2604.01733v1 (2026-04-02) โ€” **established for numeric/table domains.** + +**Claim (the strongest negative result): prompt-only LLM query rewriting (Ministral-8B rewriter, MPNet + BGE-base retrievers) produced FiQA **โˆ’9.0% nDCG@10 (p<0.001)**, TREC-COVID +5.1% (p=0.024, marginal after correction), SciFact no effect. Mechanism: vocabulary-overlap decline on FiQA; gains correlate with term standardization toward corpus vocabulary. Crucially, **an attempt to gate the rewriter achieved only AUC 0.593, with an oracle gating ceiling of ~+3 pp.** +โ†’ https://arxiv.org/html/2603.13301 (Varun Kotte, Adobe, 2026-03-02) +**Confidence: established for the negative result.** This is the strongest evidence that *"just gate your rewriter"* is currently hard to execute. + +**Claim (the efficient alternative): STORM trains 0.6Bโ€“8B models with BM25-score rewards to generate lexical expansions via reward-guided beam search. At 8B it "rivals far larger proprietary rewriters," matches/surpasses competitive LLM rewriters on TREC DL + BEIR, **retains BM25-level speed**, and zero-shots to 18 languages (MIRACL). Explicitly positioned as "infrastructure-light" โ€” no corpus re-encoding.** +โ†’ https://arxiv.org/abs/2606.10621 (2026-06-09) +**Confidence: emerging.** Directly contradicts Jina's 2025 claim that query rewriting "rules out small LMs." + +## 6.4 Self-query / structured metadata filters โ€” the quiet winner + +**Claim: LLMโ†’structured-filter translation is near-perfect on easy/medium queries and collapses on comparative/aggregate ones. F1 at optimal threshold on a Chroma-backed nutrition corpus:** + +| Difficulty | Gemini-2.0-Flash | Claude-Sonnet-4 | GPT-4o | Mistral Medium 3 | +|---|---|---|---|---| +| Easy | 0.999 | 0.999 | 0.999 | 0.999 | +| Medium | 1.000 | 1.000 | 0.990 | 0.998 | +| **Hard** | **0.445** | **0.450** | **0.396** | **0.425** | + +Verbatim: *"Even without fine-tuning, an open-source LLM such as Mistral can serve as a highly reliable metadata filter generator."* Failure mode is structural โ€” queries whose constraints exceed the metadata schema's representational scope โ€” not model capacity. +โ†’ https://arxiv.org/html/2603.09704v2 (2026-03-11) +**Confidence: established for easy/medium; emerging for the hard-tier collapse.** + +**This is the one query-planning technique where a small local model is already sufficient.** Unlike HyDE (needs generation quality) or rewriting (needs corpus-vocabulary knowledge), filter extraction is a constrained-decoding classification problem. + +**Claim: vector-DB vendors ship this as a product default. Weaviate's Query Agent auto-extracts typed filters (documented example: `IntegerPropertyFilter(property_name='price', operator=LESS_THAN, value=200.0)`), routes across collections, and returns provenance. Qdrant's 2026 engineering investment went into filtered vector search performance (ACORN analysis, 2026-07-27) โ€” which only matters if filter extraction is a mainline path.** +โ†’ https://weaviate.io/blog/query-agent (2025-03-05) ยท https://qdrant.tech/blog/ +**Confidence: established.** + +## 6.5 What to ship + +| Technique | Default? | Expected gain | Cost | +|---|---|---|---| +| Hybrid BM25+dense, RRF | **YES** | +0.051 R@5 over BM25, +0.108 over dense | 0 extra LLM calls | +| Cross-encoder rerank | **YES, unconditionally** | **+17.2 pp MRR@3** | negligible | +| Self-query metadata filters | **YES where a schema exists** | ~0.999 F1 easy/medium | 1 constrained decode | +| Index-time contextual enrichment | **YES** | +2.2 pp R@5 (T2-RAGBench); 35โ€“67% failure reduction (Anthropic) | offline only | +| Query decomposition | **Conditional** โ€” multi-hop, applied at *rerank* | +1.4 EM local 7B; +35% doc precision w/ bandit | **~2ร— latency** | +| Multi-query expansion | **NO** | โ‰ˆ0 to negative | +1 LLM call + Nร— retrieval | +| HyDE | **NO โ€” gate post-retrieval** | โˆ’7.3% (numeric) to positive (narrative) | +25โ€“60% latency on small models | + +**Two rules that generalize:** (1) **spend on the reranker before the query** โ€” every 2026 head-to-head puts reranking's ROI an order of magnitude above any query transformation; (2) **escalate, don't pre-decide** โ€” pre-retrieval routing is structurally unreliable; cheapest-first cascades win. + +--- + +# 7. ROUTING / TRIAGE + +## 7.1 Small-model routers: lexical beats semantic, and no LLM is needed + +**Claim: for RAG-*strategy* routing, TF-IDF + SVM (RBF) beats sentence embeddings and every neural option tested. On RAGRouter-Bench (7,727 queries, 4 corpora, 5 paradigms), 15 classifier ร— feature combinations, 5-fold CV:** + +| Classifier | TF-IDF Acc / F1 | MiniLM Acc / F1 | Structural Acc / F1 | +|---|---|---|---| +| **SVM (RBF)** | **93.2 / 0.928** | 90.3 / 0.897 | 79.1 / 0.774 | +| MLP | 92.7 / 0.923 | 90.1 / 0.896 | 80.3 / 0.788 | +| Logistic Reg. | 92.1 / 0.918 | 86.8 / 0.864 | 78.1 / 0.763 | +| Majority class | 52.9 / 0.231 | โ€” | โ€” | + +Token savings vs always-IterativeRAG: TF-IDF+SVM **28.1% at 0.928 F1**; perfect-label ceiling 35.2%. **Critical warning: the majority-class baseline (always route to the cheapest paradigm) achieves 60% savings at 0.231 F1 โ€” "optimising savings in isolation simply collapses routing to the cheapest paradigm."** Domain macro-F1 spread: legal 0.967, literature 0.951, Wikipedia 0.926, medical 0.803. +โ†’ https://arxiv.org/abs/2604.03455 (2026-04-03) โ€” โš ๏ธ *preprint, in-distribution only; authors flag OOD as untested.* +**Confidence: emerging, but cheap and directly actionable.** Stated reason lexical wins: dense embeddings conflate surface-similar but type-different queries. + +**Underlying benchmark: RAGRouter-Bench establishes relative token costs โ€” LLM-Only 1.0ร— / NaiveRAG 1.4ร— / GraphRAG 2.1ร— / HybridRAG 2.8ร— / IterativeRAG 3.5ร— โ€” and finds "no one-size-fits-all paradigm exists," with optimal selection driven by *queryโ€“corpus interaction* (corpus connectivity, density, intrinsic dimension, hubness), not query text alone.** +โ†’ https://arxiv.org/abs/2602.00296 (2026-01-30, rev 2026-04-04) +**Confidence: established as the reference benchmark; new and not yet replicated.** + +**Claim: generative SLMs used zero-shot as front-door routers do not meet production bars. On a 60-case benchmark: Qwen2.5-3B 0.783โ€“0.793 accuracy @ ~1,000 ms median; Phi-3.5-mini (3.8B) 0.717 @ 5,772 ms; **Qwen2.5-1.5B collapses to 0.400**; DeepSeek-V3 (671B) 0.830. Verbatim: "No model meets the standalone viability criterion (โ‰ฅ0.85 accuracy, โ‰ค2,000 ms P95)."** +โ†’ https://arxiv.org/html/2604.02367 (2026-03-26) โ€” โš ๏ธ *tiny benchmark (neff=60).* +**Confidence: emerging/weak, but directionally clear: 3B is the practical floor for *generative* routing.** + +**โš ๏ธ Direct answer on "sub-1B routers like Qwen3-0.6B": no primary 2026 source benchmarks a sub-1B *generative* model as a retrieval router.** Qwen3-0.6B's official positioning is embedding/reranking (https://huggingface.co/Qwen/Qwen3-Reranker-0.6B). What the evidence supports: **use a discriminative sub-1B router** โ€” TF-IDF+SVM (no neural net at all, 0.928 F1) or MiniLM-L6-v2 (22M params, CPU-only, 0.897 F1). + +## 7.2 Embedding-similarity routers + +- **Aurelio Labs `semantic-router`** โ€” https://github.com/aurelio-labs/semantic-router, actively maintained (2,376 commits, fetched 2026-08-08). Define `Route` objects by example utterances, encode, cosine-match, route or return `None`. **No published latency benchmarks on the repo.** *established that it exists; unsupported quantitatively.* +- **Counter-evidence:** MiniLM sentence embeddings **lose to TF-IDF by 3.1 macro-F1** on RAGRouter-Bench. *emerging.* +- RouterBench's 2024 finding that "KNN and MLP routers on sentence embeddings are competitive" (https://arxiv.org/abs/2403.12031, 2024-03) is **still the most-cited data point and has not been refreshed.** + +## 7.3 LLM-as-router +Lineage: RouteLLM (ICLR 2025), FrugalGPT (TMLR 2024), Hybrid LLM (ICLR 2024 โ€” cuts large-model calls 40% with no quality loss using a DeBERTa router). **The consistent four-year pattern: a small discriminative classifier is the right tool; an LLM router is rarely justified. Confidence: established.** + +## 7.4 When industry skips routing entirely โ€” increasingly, yes + +Three independent 2026 results converge: + +1. **Rule-based retriever routing loses to fixed hybrid.** Router rules were "named entities/dates/numbers โ†’ BM25; short or conceptual โ†’ dense; else hybrid." Failure: named entities appear in nearly every HotpotQA sub-question, so **72% of retrieval calls went to BM25**, and **hybrid-only beat the adaptive pipeline by +1.8 EM / +1.9 F1 (p<0.001)**. โ†’ https://arxiv.org/abs/2606.21553 (2026-06-19) +2. **Pre-retrieval routing for augmentation is structurally impossible.** Four ML approaches all failed; the cascade replacement gave **+0.140 quality, โˆ’31.8% latency, zero training.** โ†’ https://arxiv.org/abs/2605.27220 (2026-05-26) +3. **Simple baselines match learned routers.** *"Retrieval harm is non-negligible"*; *"router rankings vary across datasets and budgets"*; **"simple uncertainty or retrieval-score baselines often rival learned utility routers"**; *"nominal thresholds frequently miss target usage rates."* โ†’ https://arxiv.org/abs/2607.24010 (2026-07-27) + +**Where routing still pays:** *paradigm/depth* selection when paradigms have genuinely different cost profiles (1.0ร—โ€“3.5ร—), where 28.1% token savings at 0.928 F1 is real money. + +**Recommendation: skip routing for "which retriever" โ€” always hybrid. Consider routing for "which pipeline depth," implemented as a post-retrieval cascade rather than a pre-retrieval classifier. Confidence: emergingโ†’established (three independent 2026 sources agreeing).** + +## 7.5 Adaptive retrieval gating โ€” alive, but its job description changed + +**Claim (the key 2026 result): adaptive retrieval underwent a role shift โ€” it is a *noise filter* for weak models and an *efficiency optimizer* for strong ones. Across 8 backbones ร— 3 datasets (ASQA/QAMPARI/ELI5 via ALCE):** + +| Backbone | Vanilla-0 | Vanilla-10 | Rerank-10 | AdaRankLLM | Oracle | +|---|---|---|---|---|---| +| Alpaca-7B | 12.55 | 18.01 | 17.83 | **20.15** | 33.35 | +| Mistral-Instruct | 18.33 | 26.02 | 25.66 | **26.71** | 42.23 | +| Qwen2.5-7B-Inst | 19.52 | 30.00 | **30.05** | 29.42 | 41.58 | +| Qwen3-8B-NoThinking | 20.37 | **32.72** | 32.47 | 32.14 | 43.81 | +| Qwen3-8B-Thinking | 21.85 | 34.18 | **34.43** | 32.75 | 45.17 | +| GPT-4o | 34.50 | 35.26 | 33.62 | **35.60** | 49.09 | + +Verbatim: for weak models it "acts as a critical noise filter" (Alpaca/Mistral peak at k=1โ€“3 and degrade with larger sets); for strong models it "serves as an efficiency optimizer," with Qwen3-8B-Thinking's Vanilla-10 often best because explicit reasoning acts as *"a potent internal verification mechanism."* Also **"The Fallacy of Static Retrieval"** โ€” optimal k is *"highly volatile"* across task, dataset and backbone. **The oracle gap persists everywhere** (34.43 vs 45.17): *"the fundamental challenges of RAG remain far from being fully resolved."* +โ†’ https://arxiv.org/abs/2604.15621 (2026-04-17), code at github.com/USTC-StarTeam/adaptive-listwise-ranking-rag +**Confidence: established.** **Direct implication: on a reasoning-capable local model (Qwen3-8B-Thinking class), adaptive filtering buys you *tokens*, not *accuracy*.** + +**Claim: training-free gating now matches trained gates. TARG generates a short no-context draft, computes mean token entropy / top-1-vs-top-2 logit margin / small-N variance, and retrieves only above threshold. Model-agnostic, no training, no auxiliary heads. Result: matches or improves EM/F1 vs Always-RAG while **reducing retrieval 70โ€“90%**, at latency near Never-RAG.** +โ†’ https://arxiv.org/abs/2511.09803 (2025-11-12, v2 through 2026-04-14) +**Confidence: emerging. Directly implementable on a local model since you own the logits.** + +**Claim: confidence-based gating recovers 95% of the maximum accuracy gain using 58% of retrieval operations on TriviaQA. Qwen3-4B calibration improves AUROC 0.806โ†’0.879, calibration error 0.163โ†’0.034. **Core theoretical result: SFT produces well-calibrated confidence (MLE); RL (PPO/GRPO/DPO) induces overconfidence via reward exploitation.** Fix: post-RL SFT with self-distillation.** +โ†’ https://arxiv.org/abs/2603.06604 (2026-02-18) +**Confidence: emerging.** โš ๏ธ **Serious, under-appreciated warning: if you deploy an RL-trained search agent (Search-R1 lineage) *and* a logit-based retrieval gate, the RL training actively breaks the gate.** + +**Claim: adaptive RAG routers are brittle to surface query perturbations โ€” "surface query changes dramatically alter retrieval decisions."** โ†’ https://arxiv.org/abs/2604.10745 (2026-04). *emerging.* + +**Lineage status (all confirmed alive and cited in 2026 work):** Self-RAG (ICLR 2024) โ†’ FLARE (EMNLP 2023) โ†’ SKR (EMNLP 2023) โ†’ Adaptive-RAG (NAACL 2024) โ†’ CRAG (2024) โ†’ Probing-RAG (NAACL Findings 2025) โ†’ CTRLA/TARG (2025) โ†’ AdaRankLLM (2026). + +## 7.6 Routing benchmarks + +| Benchmark | URL | Date | Scope | +|---|---|---|---| +| RouterBench | https://arxiv.org/abs/2403.12031 | 2024-03 | 405k inference outcomes, LLM routing. Still the reference. | +| RouterArena | https://arxiv.org/abs/2510.00202 | 2025-09-30 | Open router leaderboard. **No 2026 update found.** | +| **RAGRouter-Bench** โญ | https://arxiv.org/abs/2602.00296 | 2026-01-30 | **First benchmark for RAG-strategy routing** | +| KARLBench | https://arxiv.org/abs/2603.05218 | 2026-03-05 | Enterprise search agents, 6 regimes | +| OrchestraBench | https://arxiv.org/abs/2608.05263 | 2026-08-05 | Multi-agent orchestration w/ failure injection | + +**Scope note: the RouterBench/RouterArena lineage is about *model* routing, not *retrieval* routing. RAGRouter-Bench (Jan 2026) is the first benchmark answering the retrieval question.** + +--- + +# 8. AGENTIC LOOP PATTERNS + +## 8.1 The four patterns, measured + +**Pattern A โ€” retrieve-then-generate.** Local 7B HotpotQA: **EM 43.1 / F1 54.0 / 546 ms.** LangChain frames this as "high control / low flexibility / fast." +**Pattern B โ€” ReAct-style search loop.** Same hardware: **EM 53.2 / F1 61.6 / 5,642 ms** at 3 steps; hybrid-only variant **EM 55.0 / F1 63.5**. โ†’ **+11.9 EM for ~10ร— latency.** Depth โ‰ฅ3 buys nothing. +โ†’ https://arxiv.org/abs/2606.21553 (2026-06-19) ยท https://docs.langchain.com/oss/python/langchain/retrieval + +**Pattern C โ€” plan-execute (plan before search).** Decompose into ordered sub-questions *before any retrieval*, so each search step is "anchored to a pre-designed sub-question instead of drifting under the influence of partially relevant documents." Validated at **3Bโ€“14B across three model families**, using a **self-bootstrapping paradigm** (small seed models generate filtered trajectories) rather than frontier distillation โ€” notably practical for self-hosting. +โ†’ https://arxiv.org/abs/2605.28354 (2026-05-27). *emerging.* + +**Pattern Cโ€ฒ โ€” parallel evidence acquisition ("search more, think less").** Replaces sequential reasoning with parallel evidence acquisition under a constrained context budget. **BrowseComp 48.6 / GAIA 75.7 / Xbench 82.0 / DeepResearch Bench 45.9**, with **โˆ’70.7% average reasoning steps on BrowseComp while improving accuracy.** +โ†’ https://arxiv.org/abs/2602.22675 (2026-02-26). *emerging; base model size not stated.* + +**Empirically-grounded ordering for a local 7Bโ€“32B: A โ‰ช B < C/Cโ€ฒ โ‰ค D, with the Aโ†’B jump by far the largest and cheapest.** + +## 8.2 RL-trained search agents + +**Lineage (all primary):** DeepRetrieval (https://arxiv.org/abs/2503.00223, 2025-02-28, **3B, 65.07% recall on publication search, beats GPT-4o**) โ†’ R1-Searcher (https://arxiv.org/abs/2503.05592, 2025-03-07) โ†’ Search-R1 (https://github.com/PeterGriffinJin/Search-R1; Llama-3.2-3B / Qwen2.5-7B bases; last notable milestone Oct 2025 โ€” **now a reference implementation, not SOTA**) โ†’ ZeroSearch (https://arxiv.org/abs/2505.04588, 2025-05-07, **7B simulation module โ‰ˆ real search engine**) โ†’ R1-Searcher++ (https://arxiv.org/abs/2505.17005, 2025-05-22). + +**2026 successors (selected):** + +| Paper | arXiv | Date | Contribution | +|---|---|---|---| +| CuSearch | 2605.11611 | 2026-05-14 | Curriculum rollout sampling; **+11.8 EM over standard GRPO** | +| Rยฒ-Searcher | 2606.28566 | 2026-06-26 | Reasoning-reflection RL, tree exploration, 7 multi-hop benchmarks | +| GraphPO | 2606.18954 | 2026-06-17 | Rollouts as DAGs, merges equivalent paths, variance reduction | +| OASES | 2604.03675 | 2026-04-04 | Co-trains search policy + state evaluator | +| Hide to Guide (SMEPO) | 2605.25198 | 2026-05-24 | Semantic masking prevents reward hacking; **+3.2 pts over GRPO** | +| **DEEPRUBRIC** | 2606.17029 | 2026-06-15 | Evidence-tree rubrics; **8B โ‰ˆ prior SOTA at ~13ร— fewer RL GPU-hours** | +| Speculate While You Reason | 2607.25816 | 2026-07-28 | Qwen3-4B/4.5B; **next-tool-call Hit@1 44โ€“49% โ†’ 61โ€“66%** | +| **MAPD** | 2607.24280 | 2026-07-27 | Distills a **style-normalized JSON protocol** (task type + plan + grounding facts), not logits. **Qwen3-1.7B 39.4%, Qwen3-4B 44.4%** across 7 QA benchmarks; "consistently outperforms competitive distillation and RL" | + +**Meta-observations:** (1) the field moved from **outcome rewards** (2025) to **credit assignment** (2026) โ€” step-level, graph-based, tree-based, rubric supervision; (2) **training cost collapsed** (~13ร— fewer GPU-hours for equal quality); (3) **local simulation is the cost story** โ€” LiteResearcher logged **73.2M tool calls (45.8M searches + 27.4M browses)** during RL, which would cost **$59Kโ€“$243K** online but **zero marginal cost** locally, with **10โ€“46ร— latency speedup** and no environmental noise (https://arxiv.org/html/2604.17931v5, 2026-07-28). **Direct validation of the self-hosted thesis.** +**Confidence: established for the lineage map; emerging for individual numbers (mostly unrefereed preprints).** + +## 8.3 The single most important benchmark result for self-hosters + +**Claim: on BrowseComp-Plus (fixed, human-verified corpus with controlled retrieval, so the retriever and the agent are disentangled): Search-R1 + BM25 scores 3.86%; GPT-5 + BM25 scores 55.9%; GPT-5 + Qwen3-Embedding-8B scores 70.1% โ€” with *fewer* search calls.** +โ†’ https://aclanthology.org/2026.acl-long.1023/ (**ACL 2026 Main**) ยท https://github.com/texttron/BrowseComp-Plus ยท original https://arxiv.org/abs/2508.06600 +**Confidence: established (peer-reviewed).** +**โ†’ Swapping the retriever moved the same agent +14.2 points. This is the strongest published evidence that for a self-hosted agent, retriever quality dominates agent scaffolding.** Fix your embedder before you build a loop. + +## 8.4 Chroma Context-1 โ€” the best-documented open artifact for this use case + +**Claim: Context-1 is a purpose-trained, open-weight (Apache 2.0) retrieval subagent. 20B derived from gpt-oss-20b, MXFP4-quantized. SFT on Kimi K2.5 trajectories โ†’ RL via CISPO (a GRPO variant), ~230 steps, 8,000+ synthetic tasks over web/finance/legal/email, curriculum from recall-weighted (16:1) to precision-weighted (4:1). A `prune_chunks` tool lets it discard irrelevant retrieved docs mid-search within a fixed 32.8k token budget. It returns ranked supporting documents to a downstream answering model, "cleanly separating search from generation."** + +| Benchmark | Context-1 (1ร—) | Context-1 (4ร—) | Frontier baselines | +|---|---|---|---| +| Web (difficulty 2+) | 0.88 | 0.97 | 0.95โ€“0.99 | +| Finance | 0.64 | 0.82 | 0.65โ€“0.90 | +| Legal | 0.89 | 0.95 | 0.90โ€“0.98 | +| Email (held out) | 0.92 | 0.98 | 0.93โ€“0.98 | +| **BrowseComp-Plus** | **0.87** | **0.96** | 0.82โ€“0.94 | + +**10ร— faster inference at frontier-comparable quality; 400โ€“500 tok/s on B200.** Behavior deltas vs base: 2.56 tool calls/turn (vs 1.52), trajectory 6.7โ†’5.2 turns, prune accuracy 0.941 (vs 0.824), parallel tool calling. Explicitly contrasts itself with Anthropic's parallel-subagent design: **a single specialized retrieval subagent paired with a frontier reasoner**, avoiding running multiple frontier models. Differs from MemGPT (external paging) and ReSum (lossy summarization) by doing **selective document-level retention without compression**, preserving evidence fidelity. +โ†’ https://www.trychroma.com/research/context-1 (2026-03-26) +**Confidence: emerging** (vendor technical report, self-reported) **but it has open weights, open data-generation code, and an Apache-2.0 license โ€” the highest-quality artifact in this survey for self-hosted purposes.** + +## 8.5 Open-weight deep-research agents at 4Bโ€“32B + +| System | Base | GAIA | BrowseComp | FRAMES | xbench-DS | +|---|---|---|---|---|---| +| LiteResearcher-4B ยน | Qwen3-4B-Thinking-2507 | **71.3** | 27.5 | **83.1** | **78.0** | +| Tongyi DeepResearch-30B-A3B ยฒ | 30.5B MoE / 3.3B active | 70.9 | โ€” | โ€” | โ€” | +| Claude-4.5-Sonnet (ref) | closed | 71.2 | โ€” | 80.7 (Claude-4) | โ€” | +| SMTL ยณ | unstated | 75.7 | 48.6 | โ€” | 82.0 | + +ยน https://arxiv.org/html/2604.17931v5 (2026-07-28), 64K context + memory compression. ยฒ https://github.com/Alibaba-NLP/DeepResearch (released 2025-09-17, **Apache-2.0**, 128K context; **no 2026 successor announced as of 2026-08-08**). ยณ https://arxiv.org/abs/2602.22675 (2026-02-26). + +**Headline: a 4B open model now matches Claude-4.5-Sonnet on GAIA-Text (71.3 vs 71.2) and beats Claude-4-Sonnet on FRAMES (83.1 vs 80.7). Confidence: emerging โ€” single-source, unrefereed, self-reported. Treat as an upper bound.** + +**โš ๏ธ Flagged unreliable โ€” do not cite:** https://arxiv.org/abs/2607.27562 reports HLE 87.3%, BrowseComp-ZH 85.3%, WebWalkerQA 91.2% โ€” implausible relative to every other 2026 result and uncorroborated. + +## 8.6 Subagent fan-out โ€” now genuinely contested + +**The canonical pro-fan-out source:** Anthropic's orchestrator-worker design, where **multi-agent (Opus 4 lead + Sonnet 4 subagents) outperformed single-agent Opus 4 by 90.2%** on a breadth-first research eval; **token usage alone explains 80% of performance variance**; agents โ‰ˆ 4ร— chat tokens, **multi-agent โ‰ˆ 15ร— chat tokens**; parallel tool calling + simultaneous subagent spawning cut research time **up to 90%**. Lesson: *"Agent-tool interfaces are as critical as human-computer interfaces."* +โ†’ https://www.anthropic.com/engineering/multi-agent-research-system (2025-06-13) +**Confidence: established as an engineering account.** โš ๏ธ 14 months old, frontier models, and **the 15ร— token multiplier is disqualifying for most self-hosted budgets.** + +**The 2026 counter-evidence:** on repository-level code QA, **plain semantic search scored 65.2% while deep agentic search scored 46.2% at >2ร— cost.** **41.8% of deep-agentic failures occurred at the plannerโ†’subagent hand-off**, and these were *"usually silent, ending in a fluent and confident answer that was wrong."* For read-only, indexable questions, *"retrieval was the stronger and cheaper option"* โ€” despite deep agentic search being *"now the preferred design in many code agents."* +โ†’ https://arxiv.org/abs/2608.01507 (2026-08-02) +**Confidence: emerging (single domain), but it is the first quantified measurement of the subagent hand-off as a failure surface and it directly contradicts the default 2025 recipe.** + +**Corroborating โ€” more search โ‰  better answers:** across six agents on BrowseComp-Plus, **"search volume correlates weakly with answer quality"**; accuracy correlates better with **cumulative retrieval recall**; useful information usually appears **early** but agents keep searching; redundant queries characterize *underperforming* agents while exploratory *reformulations* remain valuable. **Recommendation: stopping criteria based on cumulative evidence sufficiency, not query count.** +โ†’ https://arxiv.org/abs/2608.01913 (2026-08-03). *emerging, strong methodology.* + +**Production escalation architecture (named industrial deployment, Ontario Power Generation, IEEE SEGE 2026):** four deployed stages โ€” naive RAG โ†’ hybrid + rerank โ†’ agentic function-calling retrieval โ†’ deep multi-agent with code synthesis and explicit planning. Principle **PEA-CAE**: *"begin with low-cost, high-precision retrieval and escalate to full-document reads only when the expected evidence gain justifies latency and cost."* Also: *"context engineering is a more tractable and economically viable path than domain-specific fine-tuning."* โš ๏ธ No published metrics. +โ†’ https://arxiv.org/abs/2607.24791 (2026-06-28) + +## 8.7 Context engineering + +**Anthropic's four pillars:** (1) **compaction** โ€” summarize near limits, preserving "architectural decisions, unresolved bugs, and implementation details"; (2) **structured note-taking** โ€” external memory files; (3) **sub-agent architectures** โ€” a subagent burns "tens of thousands of tokens or more, but returns only a condensed, distilled summary (often 1,000โ€“2,000 tokens)"; (4) **just-in-time retrieval** via lightweight identifiers instead of pre-loading. Underpinning: **"context rot"** โ€” accuracy degrades with token count. +โ†’ https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents (2025-09-29) ยท empirical basis https://www.trychroma.com/research (Context Rot, 2025-07). *established.* + +**2026 quantified follow-ups:** selective retention + automated summarization โ†’ **91.6% task completion at โˆ’63% tokens** (https://arxiv.org/abs/2606.10209, 2026-06-08); SmoothAgent lookahead โ†’ **up to 11.9ร— TTFT reduction** (https://arxiv.org/abs/2607.00151, 2026-06-30); SIEVE structure-aware Boolean retrieval โ†’ **higher accuracy at 20.7โ€“50.6% fewer tokens** (https://arxiv.org/abs/2608.02751, 2026-08-03). *all emerging.* + +**โš ๏ธ Robustness warning that should gate any open-web deep-research deployment:** a **single misleading document raises false-conclusion adoption from 0% to 54.7%** (https://arxiv.org/abs/2607.20891, 2026-07-23); a single misleading document causes **66โ€“88 pp accuracy drops** across agents (https://arxiv.org/abs/2607.17291, 2026-07-19); research-trajectory hijacking achieves 26.4% PRISM with five injected documents (https://arxiv.org/abs/2607.04718, 2026-07-06). **Three independent 2026 sources agree on the magnitude.** Lower risk over a trusted local corpus; dominant failure mode for anything touching the open web. + +## 8.8 Recommended architecture for a self-hosted 7Bโ€“32B, by measured ROI + +**Tier 1 โ€” do first (large gain, near-zero cost):** +1. **Hybrid BM25 + dense with RRF, always. No retriever routing.** (+0.108 R@5 over dense; +1.8 EM over rule-based routing) +2. **Cross-encoder reranker, unconditionally.** (+17.2 pp MRR@3; +1.7 EM at negligible latency) +3. **Invest in the retriever before the agent.** (BM25 โ†’ Qwen3-Embedding-8B = **+14.2 points on BrowseComp-Plus**) +4. **Index-time contextual enrichment.** (offline cost only) + +**Tier 2 โ€” the loop:** +5. **Iterative retrieve-reason loop capped at 2โ€“3 steps.** Step 1โ†’2 is +7.1 EM; 3โ†’5 is noise. +6. **Terminate on cumulative evidence sufficiency, not query count.** +7. **Plan-first over pure ReAct** if you can afford one planning call โ€” validated at 3Bโ€“14B with self-bootstrapped training requiring no frontier teacher. + +**Tier 3 โ€” query planning, selectively:** self-query filters ON; decomposition for multi-hop applied **at the rerank stage**; HyDE OFF by default, wired as a post-retrieval fallback; multi-query OFF (use a STORM-style 0.6Bโ€“8B BM25-reward expander instead). + +**Tier 4 โ€” adaptive gating (cost-motivated only):** TARG-style prefix-logit gating gives **70โ€“90% retrieval reduction at matched EM/F1** with no training โ€” essentially free when you own the logits. But on a Qwen3-8B-Thinking-class model it buys **tokens, not accuracy**. โš ๏ธ **Do not combine an RL-trained search agent with a logit-confidence gate without post-RL SFT recalibration.** + +**Tier 5 โ€” subagents:** prefer **one specialized retrieval subagent + one reasoner** over parallel fan-out (Chroma Context-1's explicit design choice, Apache-2.0, 20B, 10ร— speedup, 0.87โ€“0.96 BrowseComp-Plus). If you do fan out, **instrument the plannerโ†’subagent hand-off** โ€” 41.8% of failures, silently. **Self-editing context (prune mid-search under a fixed ~32.8k budget) is the 2026 replacement for lossy summarization.** + +**What NOT to do:** always-on HyDE; always-on multi-query; rule-based retriever routing on entity heuristics; loops deeper than 3; pre-retrieval classifiers for "do I need augmentation"; optimizing a router on cost savings alone (60% savings at 0.231 F1 is the degenerate solution). + +--- + +# 9. VERIFICATION / GROUNDING + CITATIONS + +## 9.1 The 2024 baseline is now the floor + +**Claim: MiniCheck's original result โ€” LLM-AggreFact balanced accuracy without per-dataset threshold tuning: AlignScore 70.4, MiniCheck-RoBERTa 72.7, MiniCheck-DeBERTa 72.6, MiniCheck-Flan-T5 74.7, GPT-4 75.3. Inference cost on the 13K test set: AlignScore $0.20, MiniCheck-FT5 $0.24, GPT-4 $107 (~400ร— cheaper).** +โ†’ https://arxiv.org/html/2404.10774v1 (EMNLP 2024). **established, but a 2024 result โ€” the floor, not the SOTA.** + +**Claim: current LLM-AggreFact leaderboard top-5 by average BAcc: Bespoke-MiniCheck-7B 77.4, Claude-3.5 Sonnet 77.2, IBM Granite Guardian 3.3 (8B) 76.5, Mistral-Large-2 (123B) 76.5, gpt-4o 75.9. 39 models ร— 11 datasets.** +โ†’ https://llm-aggrefact.github.io/ (fetched Aug 2026) +**Confidence: established for the numbers; contested as a *current* signal โ€” the page shows no 2026 refresh and the newest datable entry is Aug 2025.** + +**Claim: the MiniCheck project is dormant.** Newest repo news is Sep 2024. Models: MiniCheck-RoBERTa-Large 355M, MiniCheck-DeBERTa-v3-Large 434M, MiniCheck-Flan-T5-Large 770M, Bespoke-MiniCheck-7B. Repo Apache-2.0, but **Bespoke-MiniCheck-7B requires contacting company@bespokelabs.ai for commercial licensing.** Throughput: 29,000 instances in ~30 min on one A6000 48GB with prefix caching. Operates at (document, single sentence) granularity โ€” you must sentence-split claims yourself. +โ†’ https://github.com/Liyan06/MiniCheck. *established.* + +## 9.2 The 2026 successor: a 1B model beat the 7B SOTA + +**Claim: ThinknCheck โ€” a 1B-parameter Gemma3-1B, 4-bit quantized verifier that emits a short structured rationale then a binary verdict โ€” reaches 78.1 BAcc on LLM-AggreFact, beating Bespoke-MiniCheck-7B (77.4) with 7ร— fewer parameters. On SciFact it hits 64.7 BAcc, +14.7 absolute over MiniCheck-7B. Ablating the reasoning step collapses it to 57.5 BAcc โ€” the rationale is doing the work, not the fine-tune. Trained on LLMAggreFact-Think (24.1k reasoning-augmented examples). License CC0.** +โ†’ https://arxiv.org/abs/2604.01652 (Rao, Han, Callison-Burch, UPenn; 2026-04-02) +**Confidence: emerging** (single paper) **but this is the single most important ยง9 change of the year: a 4-bit 1B verifier now beats the 7B 2024 SOTA, which makes per-answer grounding verification cheap enough to always run.** + +**Claim: Paladin-mini (3.8B) reports avg BACC 79.31% vs 77.86% prior SOTA. โš ๏ธ Licensed CC BY-NC-ND 4.0 โ€” non-commercial, no derivatives.** +โ†’ https://arxiv.org/abs/2506.20384 (2025-06-25). *emerging.* + +**Claim: Granite Guardian 3.3 8B (IBM, 2025-08-01, Apache 2.0) is the best *permissively licensed* off-the-shelf option on the public leaderboard: LLM-AggreFact avg BAcc 0.761 (AggreFact-CNN 0.669, REVEAL 0.894), and it also covers context relevance, answer relevance, jailbreak, and function-calling hallucination in one model. A 38M variant exists for low latency.** +โ†’ https://huggingface.co/ibm-granite/granite-guardian-3.3-8b. *established.* + +## 9.3 Span-level detection โ€” where 2026 moved, and a nasty surprise + +**Claim: LettuceDetect (ModernBERT token classification) โ€” RAGTruth example-level F1 79.22%, +14.8% over Luna, ~30ร— smaller than the best models, 30โ€“60 examples/sec on a single GPU.** +โ†’ https://arxiv.org/abs/2502.17125 (2025-02-24). *established; a 2025 result, now beaten, but still the best latency/quality ratio in the encoder class.* + +**Claim (the surprise): a fine-tuned Qwen3.5-2B span detector reaches 0.689 span-F1 on a unified benchmark spanning code, tool output, structured docs and NL RAG โ€” versus **LettuceDetect-large at 0.17** and the **strongest zero-shot LLM judges at โ‰ค0.22**. It stays competitive on classic NL benchmarks: 81.8 RAGTruth example-F1, 0.724 English PsiloQA IoU, 0.60 span-F1 on code-agent output. CC BY 4.0.** +โ†’ https://arxiv.org/abs/2607.00895 (2026-07-01) +**Confidence: emerging.** **The headline finding matters enormously for agentic retrieval: encoder-era span detectors and zero-shot LLM judges both collapse (0.17โ€“0.22 span-F1) once the "context" is code or tool output rather than prose. If your agent verifies tool outputs, off-the-shelf 2025 detectors will not work.** + +**Claim: GASP โ€” grounding-sensitivity-by-perturbation, three instruction-tuned scorers at 0.5Bโ€“1.7B, RAGTruth ~0.73 response-level AUC / ~0.67 span-level AUC, competitive with entailment verifiers without a dedicated verifier model.** +โ†’ https://arxiv.org/abs/2607.04223 (2026-07-05). *emerging.* + +**Claim: full-document (32K-token) verification substantially improves detection of unsupported responses versus chunk-truncated validation, because supporting evidence is frequently outside the truncated passage.** +โ†’ https://arxiv.org/abs/2603.23508 (2026-03-04). *emerging.* **Practical implication: verifying against the retrieved chunk alone systematically under-detects.** + +## 9.4 LLM-as-judge reliability โ€” partially rehabilitated, with a metric bug + +**Claim: a plain LLM-as-a-Judge baseline performs competitively against specialized SOTA hallucination detectors on TRIVIA+ (long-context RAG); current detectors have "ample room" for improvement on RAG-style benchmarks; label noise materially degrades measured detection performance.** +โ†’ https://arxiv.org/abs/2605.11330 (2026-05-11, **ACL 2026 main**). *emerging, and directly contested against the "small NLI model beats the judge" framing.* + +**Claim (standing caution): five SOTA factuality metrics across 11 datasets are inconsistent with each other and often misestimate system-level performance; systematically weak on highly paraphrased outputs and on outputs referencing distant source sections. Recommendation: manually validate any factuality metric in your own domain.** +โ†’ https://arxiv.org/abs/2501.14883 (Godbole & Jia, 2025-01-24). **established; a 2025 result but nothing in 2026 overturns it.** + +**Claim (the metric bug): faithfulness metrics measure only *precision* and ignore *recall/coverage*, so a system can score near-perfectly by saying almost nothing. On a 7,253-instance multilingual benchmark, the most *precise* frontier model covered under half the relevant facts and **ranks last by F1**; fine-tuned 1Bโ€“7B models reached ~0.98 F1, beating all zero-shot frontier systems.** +โ†’ https://arxiv.org/abs/2606.09376 (2026-06-08, rev 2026-06-16) +**Confidence: emerging.** **If you only gate on "is every claim supported," you will train your system to abstain. Pair precision with a coverage metric.** + +**Claim: Cleanlab published a direct head-to-head of LLM-as-a-Judge vs Prometheus, Lynx, HHEM and TLM across six RAG applications, all reference-free.** +โ†’ https://arxiv.org/abs/2503.21157 (2025-03-27). **established that the comparison exists; โš ๏ธ I could not extract the numeric table โ€” abstract has no numbers, HTML 404s, PDF unparseable. Treat the ranking as unverified.** + +## 9.5 Claim decomposition + +- **Standard pipeline:** decompose into atomic claims โ†’ retrieve evidence โ†’ verify each. Known failure: strictly atomic facts strip the context needed to verify them; "molecular facts" / decontextualization is the standard fix. โ†’ https://arxiv.org/pdf/2406.20079 (2024-06). *established.* +- **Faithfulness scores are sensitive to the decomposition method itself** โ€” DecMetrics proposes structured scoring of decomposition quality as a prerequisite. โ†’ https://arxiv.org/pdf/2509.04483 (2025-09). *emerging.* +- **TriQua (2026-08-05)** resolves the granularity/context trade-off: simple claims โ†’ triples, complex claims โ†’ **hyperrelational facts with qualifiers**. TriQuaScore correlates strongly with human factuality annotations, outperforms existing decomposition frameworks, and localizes errors per-triple/per-qualifier. โ†’ https://arxiv.org/abs/2608.05228. *emerging.* + +## 9.6 Citations + +**Claim (the key architectural finding): post-hoc citation (P-Cite) achieves high coverage with competitive correctness and moderate latency; generation-time citation (G-Cite) prioritizes precision at the cost of coverage and speed. Across four attribution datasets, **retrieval quality is the main driver of attribution quality in both paradigms.** Recommendation: P-Cite-first for high-stakes domains; G-Cite only for precision-critical strict claim verification.** +โ†’ https://arxiv.org/abs/2509.21557 (2025-09-25, rev 2025-12-18, NeurIPS 2025 LLM-Eval Workshop). *emerging.* + +**Claim: deep-research agents cite badly. Link validity >94%, topical relevance >80%, but **factual support only 39โ€“77%**. Factual accuracy drops **~42% on average as tool calls scale from 2 to 150**. **Fewer than half of open-source models successfully produce cited reports one-shot.** +โ†’ https://arxiv.org/abs/2605.06635 (2026-05-07) +**Confidence: emerging.** **For a local model, citation generation is a separate capability you must verify, not something you get by prompting.** + +## 9.7 Vectara HHEM lineage + +- HHEM-2.1-**open** is the open-weight version; the leaderboard now runs on commercial **HHEM-2.3** and **FaithJudge**. โ†’ https://arxiv.org/abs/2505.04847 (EMNLP 2025 Industry). *established.* +- **The open version is materially weaker, especially at longer premises:** TofuEval-MediaSum BAcc **78.48 (2.3) vs 71.57 (2.1-open)**; RAGTruth-Summary recall **min 0.758 (2.3) vs max 0.377 (2.1-open)**. โ†’ https://www.vectara.com/blog/hallucination-detection-commercial-vs-open-source-a-deep-dive. *established (vendor self-report; direction credible, magnitude vendor-favorable).* **For a self-hosted stack this is a reason to prefer ThinknCheck / Granite Guardian / LettuceDetect over HHEM-2.1-open.** +- Next-gen leaderboard (2025-11-19) expanded to **7,700+ articles, up to 32,000 tokens**, 10 domains. Under the harder benchmark, hallucination rates rose sharply: Gemini-2.5-flash-lite leads at **3.3%**, Claude Sonnet 4.5 **>10%**. Feb 5 2026 update covers 80+ models. *established.* + +--- + +# 10. CONTEXT COMPRESSION + +## 10.1 Provence โ€” the one variant where the economics work + +**Claim: Provence (Naver Labs Europe, ICLR 2025) formulates context pruning as **sequence labeling unified with reranking**, dynamically detects how much to prune per context, works out-of-the-box across domains, and achieves "negligible to no drop in performanceโ€ฆ at almost no cost in a standard RAG pipeline" โ€” because pruning is folded into the reranker you already run. DeBERTa-v3 trained on Llama-3-8B synthetic sentence-relevance labels, plus self-distillation to preserve reranking.** +โ†’ https://arxiv.org/abs/2501.16214 (2025-01-27) ยท https://huggingface.co/blog/nadiinchi/provence ยท `naver/provence-reranker-debertav3-v1` +**Confidence: established.** **The "zero marginal cost" property is the whole argument** โ€” it sidesteps the latency objection that kills LLMLingua. + +**Claim: XProvence (ECIR 2026) extends this to 16 trained languages generalizing to 100+, "minimal-to-no performance degradation," outperforming strong baselines on four multilingual QA benchmarks. Weights: `naver/xprovence-reranker-bgem3-v2`.** +โ†’ https://arxiv.org/abs/2601.18886 (2026-01-26). *emerging.* + +**Claim: OpenProvence is a fully MIT-licensed ModernBERT reimplementation at 30M / 130M / 149M / 310M, English + Japanese. Large @ threshold 0.10 on MLDR-English: **93.10% Has-Answer with 94.38% positive-passage compression and 99.90% negative-passage compression**, matching the `naver/provence` baseline. Weights, training code, inference code and dataset tooling all MIT.** +โ†’ https://github.com/hotchpotch/open_provence +**Confidence: emerging** (single-author, self-reported) **but this is the one to actually ship in a self-hosted stack** โ€” 30Mโ€“310M, MIT, and it replaces a pipeline stage rather than adding one. + +## 10.2 LLMLingua lineage โ€” effectively frozen + +**Claim: LLMLingua-2 offers 3ร—โ€“6ร— speedup over LLMLingua via GPT-4-distilled token classification; the family claims up to 20ร— compression with minimal loss; LongLLMLingua reports up to +21.4% at 1/4 the tokens. Integrated into LangChain, LlamaIndex, Prompt Flow. MIT, 6.5k stars. **But the repo's newest news item is Dec 2024** โ€” no new compressor in ~20 months.** +โ†’ https://github.com/microsoft/LLMLingua ยท https://arxiv.org/abs/2310.06839 +**Confidence: established as a claim; treat LLMLingua-2 (2024) as the terminal version.** + +**Claim: LLMLingua-2 does not transfer to diffusion LLMs; high semantic preservation does not guarantee stable downstream behavior, and mathematical reasoning degrades substantially.** โ†’ https://arxiv.org/pdf/2605.17932 (2026-05-18). *emerging.* + +## 10.3 The decisive 2026 evidence AGAINST compression + +**Claim (the latency argument โ€” best single answer to "is compression worth it in 2026"): across thousands of runs on 30,000 queries, multiple open-source LLMs, and three GPU classes, LLMLingua yields **at most ~18% end-to-end speedup, and only when prompt length, compression ratio, and hardware all align.** Outside that operating window, **compression overhead dominates and cancels the decoding gains entirely.** Output quality was statistically unchanged across summarization, code generation and QA. The authors released an open-source profiler predicting the latency break-even point per model/hardware pair. One genuine win: compression cut memory enough to migrate workloads from datacenter GPUs to commodity hardware with minimal latency penalty.** +โ†’ https://arxiv.org/abs/2604.02985 (Kummer et al., **ECIR 2026 full paper**, 2026-04-03) +**Confidence: established** (large-scale, systems-level, peer-reviewed). + +**Claim (the evaluation-validity argument): fixed compression can raise average accuracy while hiding reader upgrades and reversing model rankings. Across 20 readers ร— 10 domain-method configurations on 4 QA + 1 summarization benchmark: **compression benefit shrinks as the reader gets stronger (significant in 9/10 settings, p<0.05)**. Generic summarization **reversed 31% of pairwise model rankings on LongMemEval-S**. A HotpotQA compressor **obscured 80% of the gain from upgrading Qwen-7B โ†’ GPT-4.1-mini**. Mechanism: compression helps weak readers by removing noise they cannot filter, and hurts strong readers by removing detail they could have used. Toolkit `ragscale`, built on 177,000 compression transitions.** +โ†’ https://arxiv.org/abs/2606.21807 (2026-06-20) +**Confidence: emerging but methodologically strong.** **This is precisely what "cheap long-context local models in 2026" changed: a compressor tuned when your reader was weak is now actively costing you accuracy AND masking the benefit of every model upgrade.** + +**Claim (the structural-failure argument): hard compressors score units independently, so they split dependent evidence pairs โ€” "referential dangling." At compression ratio 0.30, one system leaves the answer path incomplete in **34โ€“54%** of bridge examples; across six hard compressors on HotpotQA, **dangling rates reach 60%**. Reinserting the missing supporting paragraphs recovers **+29โ€“34 percentage points**; a trained restoration classifier recovers +4.7 points at the same ratio.** +โ†’ https://arxiv.org/abs/2608.04569 (2026-08-05). *emerging.* **Multi-hop RAG is where hard compression breaks.** + +**Claim (the baselines-were-wrong argument): in soft/embedding context compression, **mean pooling and a simple bidirectional compression-token variant outperform the widely used causal compression approaches.** Released BenchPress, a standardized suite spanning model scales, datasets and ratios from <1K to 8K tokens.** +โ†’ https://arxiv.org/abs/2510.20797 (2025-10-23, **v2 2026-05-10**) ยท https://github.com/lil-lab/benchpress +**Confidence: emerging.** Much of the learned-soft-compression literature was beating weak baselines. + +**Claim (decision-quality): LLM compression alters downstream *decisions* through decontextualization; fluent, factually plausible compressions still change decision-relevant judgments in financial analysis.** โ†’ https://arxiv.org/pdf/2606.29251 (2026-06-28). *emerging.* **Faithfulness-preserving โ‰  decision-preserving.** + +**Claim (agent control): under compression, degradation surfaces as **tool-execution failures, not lower token counts**. At 35% retained context: section-based 47.0% success, obligation-aware 39.0%, generic rewriting **19.9%**.** โ†’ https://arxiv.org/pdf/2608.01056 (2026-08-02). *emerging.* + +## 10.4 The 2026 evidence FOR compression + +- **CORE-RAG** โ€” performance-driven compression trained with task success as feedback. At a **3% compression ratio it improves average Exact Match by +3.3 points over using full documents.** ICML 2026. โ†’ https://arxiv.org/abs/2508.19282 (v4 2026-05-28). *emerging.* Note: learned-for-the-task, not a drop-in. +- On repository-level code tasks at **4ร— compression**, continuous-latent-vector methods **surpass full context by +28.3% BLEU** โ€” "compression filters noise rather than just truncating." โ†’ https://arxiv.org/pdf/2604.13725 (2026-04-15). *emerging.* +- **Telegraph English** โ€” structured rewriting at ~**50% token reduction preserves 99.1% key-fact accuracy** with GPT-4.1, and **beats LLMLingua-2 by up to 11pp on fine-detail tasks in smaller models.** โ†’ https://arxiv.org/pdf/2605.04426 (2026-05-06). *emerging.* +- **Evolved linguistic rules** match advanced prompt-compression strategies **with no LM forward pass at deployment** โ€” zero compression latency, removing the overhead that kills LLMLingua economics. โ†’ https://arxiv.org/pdf/2607.25335 (2026-07-29). *emerging.* +- **Tool-schema compression**, not document compression, is where agentic RAG wins: **44โ€“50% schema token savings and +20.5pp average exact-match recovery** in constrained contexts. โ†’ https://arxiv.org/abs/2605.26165 (2026-05). *emerging, under-appreciated โ€” in an agentic loop, tool schemas often outweigh retrieved text.* +- **RECOMP** remains the canonical baseline Provence/DSLR compare against; **not independently re-verified this session** โ€” treat as superseded in the pruning-reranker line. + +## 10.5 Verdict + +- **Prune, don't compress.** Sentence-level extractive pruning folded into the reranker (**Provence / OpenProvence / XProvence**) is the only variant where the latency argument is not fatal. +- **Do not ship a token-level LLM compressor without profiling.** ECIR 2026 measured โ‰ค18% best case, net-negative outside a narrow window; upstream quiet since Dec 2024. +- **Do not use a fixed compression ratio.** It hides reader upgrades and reverses rankings. +- **Never hard-compress multi-hop contexts** without a dangling-restoration step. +- **Re-run your compression ablation every time you upgrade the local reader.** + +--- + +# 11. MEMORY + +## 11.1 The benchmark situation is bad, and now independently documented + +**Claim: LOCOMO โ€” the field's standard benchmark โ€” is broadly discredited. Objections: conversations average only 16kโ€“26k tokens (inside modern context windows, so they do not stress memory at all); a plain full-context baseline beats the memory systems in Mem0's own reported results; plus data defects (missing ground-truth answers, multimodal errors, incorrect speaker attribution, ambiguous questions).** +โ†’ https://blog.getzep.com/lies-damn-lies-statistics-is-mem0-really-sota-in-agent-memory/ (2025-05-06) +**Confidence: contested (vendor-on-vendor), but the token-count and full-context-baseline objections are structural and verifiable.** + +**Claim: a plain filesystem agent with no memory tool at all scored 74.0% on LoCoMo with GPT-4o-mini, beating Mem0's reported 68.5% best variant.** +โ†’ https://www.letta.com/blog/benchmarking-ai-agent-memory (2025-08-12) +**Confidence: contested (vendor), but it is the cleanest single refutation: the null baseline wins.** + +**Claim (the first independent controlled ablation, and the most important memory finding of the year): MemDelta ran a controlled protocol against RAG and full-context baselines on LongMemEval-S (500 questions, 50+ sessions):** +- **Mem0 beats MiniLM-RAG by +11pp but LOSES to cloud-RAG by 1.2pp.** +- On 2 of 6 question types, **Mem0 matches cloud RAG (72.7% vs 73.9%, p = 1.0) at 50ร— the cost.** +- **Agent self-memory underperforms basic retrieval: 42% vs 47%.** +- **Swapping only the embedding model changes accuracy by +6.2pp at n=500 (p=0.004)** โ€” enough to reverse which system "wins." +- Reader model flips rankings: **Gemini gains +14pp from full context while Sonnet gains +31pp from RAG.** +โ†’ https://arxiv.org/abs/2606.29914 (2026-06-29). *emerging.* + +**Claim: metric choice alone swings LOCOMO โ€” a controlled ablation found a 27.5-point discrepancy between strict token-F1 and LLM-as-judge on the same outputs.** โ†’ https://arxiv.org/abs/2606.22030 (v2 2026-06-20). *emerging.* **Any LOCOMO number quoted without naming the scorer is uninterpretable.** + +**Claim: LOCOMO is saturated. In Junโ€“Aug 2026 alone, 25+ distinct memory systems report LOCOMO results, several claiming 93%+ (Maximem Synap 93.2%, MemStack 93.59%, ABot-AgentOS 87.5%).** โ†’ arXiv API sweep, fetched 2026-08-08. *established as an observation.* + +**2026 successor benchmarks:** + +| Benchmark | What it adds | Source | Date | +|---|---|---|---| +| **MemoryArena** | Interdependent multi-session *agentic* tasks. Finds agents **near-saturated on LoCoMo perform poorly** here | https://arxiv.org/abs/2602.16313 | 2026-02-18 | +| **Letta Context-Bench** | Filesystem + Skills suites, live leaderboard. Top: GPT-5.2-Codex-xhigh 93% filesystem; Claude Sonnet 4.5 72% skill-use | https://leaderboard.letta.com/ | refreshed 2026-03-13 | +| **Supersede** | Isolates **fact-supersession** failure (stale memory not overwritten) + RL environment | https://arxiv.org/abs/2606.27472 | 2026-06-25 | +| LoCoMo-Contam | Memory poisoning variant | https://arxiv.org/abs/2607.22962 | 2026-07-25 | +| LoCoMo Temporal Plus | Conflict-heavy "ghost memory" variant | https://arxiv.org/abs/2607.01935 | 2026-07-02 | + +**MemoryArena and Context-Bench are the two worth actually running.** + +**LongMemEval** (https://arxiv.org/abs/2410.10813, 2024-10-14): 500 questions over scalable chat histories; **commercial assistants and long-context LLMs show a ~30% accuracy drop** across sustained interactions. *established.* + +## 11.2 The open implementations + +**Mem0** โ€” https://github.com/mem0ai/mem0 ยท https://arxiv.org/abs/2504.19413 +62.8k stars, **Apache 2.0**, Python + TS, three deployment modes (library / self-hosted Docker / cloud), provider-pluggable. Paper claims (2025-04-28): **+26% relative LLM-as-judge over OpenAI memory, 91% lower p95 latency, >90% token cost savings**. Repo claims (Apr 2026): 92.5 LoCoMo, 94.4 LongMemEval. **established as claims; contested as results** โ€” see MemDelta/Letta/Zep above. + +**Zep / Graphiti** โ€” https://arxiv.org/abs/2501.13956 ยท https://github.com/getzep/graphiti +Paper (2025-01-20): DMR 94.8% vs MemGPT 93.4%; **LongMemEval up to +18.5% accuracy and 90% latency reduction**. *established as a claim; vendor-authored.* Graphiti: **29.7k stars, Apache 2.0**, temporal/bitemporal knowledge graph with full provenance. Backends: Neo4j 5.26+, **FalkorDB 1.1.2+ including an embedded "Lite" variant**, Amazon Neptune + OpenSearch, Kuzu 0.11.2 (**deprecated โ€” upstream unmaintained**). **Runs fully local**: FalkorDB Lite (embedded, Python 3.12+) + Ollama / vLLM / llama.cpp / LM Studio via OpenAI-compatible endpoints. **This is the strongest fully-local option.** + +**Letta / MemGPT lineage** โ€” https://github.com/letta-ai/letta +24.2k stars, Apache 2.0. **Architecture shift in the last 12 months:** the `letta` repo is now the **legacy V1 API server**; active development moved to `letta-ai/letta-code`, self-hosting via the App Server, with the SDK reorganized around **skills and subagents**. Letta's position: memory tooling matters less than agentic context management โ€” hence Context-Bench rather than LoCoMo. *established / emerging.* + +**LangMem** โ€” https://github.com/langchain-ai/langmem. 1.6k stars, **MIT**. Hot-path memory tools + background consolidation; usable with any store plus native LangGraph integration. *established*, but note the thin ecosystem (1.6k vs Mem0's 62.8k). + +**GPTCache** โ€” https://github.com/zilliztech/GPTCache. 8.1k stars, **but the latest release is v0.1.44 (2024-08-01)**, and maintainers state they "no longer add support for new API or models." +**Confidence: established โ€” GPTCache is dormant. Do not adopt it for a 2026 build.** + +## 11.3 Semantic caching โ€” 2026 research + +| Finding | Number | Source | Date | +|---|---|---|---| +| Semantic caches are **attackable** โ€” embedding-similarity collision hijacks cached answers | **86% hijacking hit rate** | https://arxiv.org/abs/2601.23088 | 2026-01-30 | +| Retrieval-quality metrics mislead for caches (calibration mismatch); proposes P-CHR AUC | โ€” | https://arxiv.org/abs/2606.19719 | 2026-06-19 | +| Freshness-aware caching for open-web RAG | **97% search-API savings at 0.1% stale-error rate** | https://arxiv.org/abs/2607.04281 | 2026-07-05 | +| Temporal semantic cache | **30.6ร— speedup on hits**, fails on time/parameter-dependent outputs | https://arxiv.org/abs/2605.20630 | 2026-05-20 | +| Production NL-to-code multi-agent cache | **67% hit rate, 40โ€“60% token reduction** | https://arxiv.org/abs/2601.11687 | 2026-01-16 | +| LRU/LFU **provably underperform** on semantic workloads; SOLAR | **5โ€“75% relative improvement** | https://arxiv.org/abs/2607.00394 | 2026-07-01 | +| Optimal offline semantic-cache policy is **NP-hard** | โ€” | https://arxiv.org/abs/2603.03301 | 2026-02-07 | + +*all emerging.* **Consistent theme: hit rate is the wrong headline metric; calibration, freshness, and collision-resistance decide whether a semantic cache is safe.** + +## 11.4 What to ship + +- **Graphiti + FalkorDB Lite + Ollama** if you need temporal/entity memory with provenance, Apache 2.0, and no network egress. The only mature genuinely-embeddable option. +- **Plain RAG over a session transcript store as the baseline you must beat.** Cloud-RAG beat Mem0 on LongMemEval-S; a filesystem agent beat Mem0 on LoCoMo. **Build the null baseline first.** +- **Mem0** for the biggest ecosystem, if you accept self-reported benchmarks. Apache 2.0, self-hostable. +- **LangMem** only if already on LangGraph. +- **Skip GPTCache.** Build a semantic cache on your existing vector store and instrument calibration + staleness, not hit rate. +- **Fix your embedding model before comparing memory systems** โ€” a bare embedding swap moves accuracy 6.2pp, larger than most claimed memory-system deltas. + +--- + +# 12. EVALUATION + +## 12.1 RAGAS in 2026 โ€” maintained, and sharply criticized + +**Claim: RAGAS is actively developed โ€” 15.2k stars, 1,147 commits, currently v0.4.x. v0.4.0 was a substantial breaking migration to a collections-based metrics system with `instructor.from_provider` universal provider support; v0.4.3 added a DSPy MIPROv2 prompt optimizer. Still ships test-data generation and Aspect Critique / `DiscreteMetric`.** +โ†’ https://github.com/explodinggradients/ragas (fetched Aug 2026). *established that it is maintained; **release-date precision is low** โ€” GitHub shows month/day only, most consistent with Dec 2025 / Jan 2026.* + +**Claim (the sharpest criticism of the year): under corpus evolution, the best RAGAS metric reaches only **F1 0.570** against ground truth, versus 0.927โ€“1.000 for metamorphic testing โ€” i.e. RAGAS fails to detect faults introduced when the knowledge base changes.** +โ†’ https://arxiv.org/abs/2607.26843 (2026-07-29). *emerging.* + +Supporting criticism: "RAGAs-based faithfulness shows limited reliability" on long structured academic documents (https://arxiv.org/abs/2607.01852, 2026-07-02); an alternative comparative scorer (DICE) reports 85.7% agreement with human experts, "substantially outperforming RAGAS" (https://arxiv.org/abs/2512.22629, 2025-12-27, single-paper self-favorable). A direct applied comparison of Ragas / DeepEval / RAGChecker / Opik against human annotators exists (https://arxiv.org/abs/2607.07302, 2026-07-08) but the correlation numbers are not in the abstract. + +**Verdict: still recommended, but only as a *relative regression signal* inside a fixed corpus and fixed judge. Do not treat RAGAS faithfulness as an absolute score, and do not rely on it to catch faults when the corpus changes.** + +## 12.2 Open judge models โ€” nothing new shipped in 12 months + +| Model | Size | License | Measured | Date | +|---|---|---|---|---| +| **Atla Selene 1 Mini** | 8B (Llama-3.1-8B) | **Apache 2.0** | #1 8B generative model on RewardBench; beats GPT-4o on RewardBench, EvalBiasBench, AutoJ; 11 benchmarks / 3 task types; 128K ctx | 2025-01-27 | +| Atla Selene 1 | 70B | โ€” | larger sibling | 2025-01 | +| **Patronus Lynx** | 8B / 70B | **CC-BY-NC-4.0 โ€” non-commercial** | HaluBench **8B 82.9%, 70B 87.4%** vs GPT-4-Turbo 85.0%; GGUF quants for llama.cpp/Ollama/LM Studio | 2024-07 | +| GLIDER (Patronus) | 3B | open | **91.3% human agreement** on arbitrary criteria, 685 domains | 2024-12-18 | +| Prometheus lineage | โ€” | open | Prometheus-Vision: highest human correlation among open VLM evaluators | 2024-01 | +| Judge's Verdict (meta) | โ€” | โ€” | Benchmarked **43 open models; 27 achieve "Tier 1"** human-like judgment (Cohen's ฮบ) | 2025-10-10 | +| Counsel (meta-eval) | โ€” | โ€” | Strongest **open-weight** judge reaches **~88% location agreement** with humans | 2026-06-19 | + +Sources: https://huggingface.co/AtlaAI/Selene-1-Mini-Llama-3.1-8B ยท https://arxiv.org/abs/2501.17195 ยท https://huggingface.co/PatronusAI/Llama-3-Patronus-Lynx-8B-Instruct ยท https://arxiv.org/abs/2407.08488 ยท https://arxiv.org/abs/2412.14140 ยท https://arxiv.org/abs/2510.09738 ยท https://arxiv.org/abs/2606.21627 + +**Flags:** **Lynx is CC-BY-NC โ€” do not ship it commercially.** **Selene 1 Mini (Apache 2.0) is the correct default open judge for a commercial self-hosted stack.** **Prometheus, JudgeLM, and Flow Judge are all 2024-vintage with no 2026 successor found** (โš ๏ธ Flow Judge and JudgeLM status unverified). The 2026 direction is not "a bigger judge" but **judge alignment tooling** (ยง12.4). + +โš ๏ธ LLM-as-judge has a **backdoor/poisoning attack surface** in RAG evaluation โ†’ https://arxiv.org/abs/2503.00596 (BadJudge, 2025-03-01). *emerging.* + +## 12.3 Frameworks for a small local team + +| Tool | License | Scale | State (Aug 2026) | +|---|---|---|---| +| **promptfoo** | MIT | 24.1k โ˜… | **Evals run 100% locally โ€” prompts never leave the machine.** Ollama + all major providers, CI/CD + PR review, red-teaming. **Acquired by OpenAI, announced 2026-03-09; OSS continues under Ian Webster & Michael D'Angelo.** Node โ‰ฅ22.22 | +| **Inspect AI** | MIT | 2.5k โ˜… | UK AI Security Institute. 7,013 commits, **200+ prebuilt evals**, model-graded scoring, tool use, multi-turn. Most auditable; smallest RAG-specific surface | +| **DeepEval** | โ€” | โ€” | v4.x (v4.1.5, Jul 29). **2026 direction is agentic**: task-completion, tool-correctness, `AgentLoopDetectionMetric`, coding-agent eval harness, TUI trace inspection, Arena-GEval | +| **TruLens** | MIT (Snowflake) | 3.5k โ˜… | **v2.12.0, 2026-08-06** (2.11.0 Aug 4, 2.10.0 Jul 28, 2.9.0 Jul 23 โ€” rapid cadence). OTel-native. **2026 additions are all judge-quality tooling**: `AlignmentReport`, `CrossModelAlignment`, `Jury`, `CriteriaABTest`, `ScoreDistributionAnalyzer`, `GoldenSetGenerator`, `FewShotOptimizer`, plus `citation_accuracy` / `citation_attribution` | +| **Arize Phoenix** | **Elastic License 2.0 โ€” NOT OSI-permissive** | 11k+ โ˜… | OTel tracing, `arize-phoenix-evals` with **published benchmarks of the evaluators themselves**, datasets/experiments, prompt playground. Self-hostable Docker/K8s/Helm | +| **Braintrust autoevals** | MIT | 994 โ˜… | Focused scorer library: context precision/relevancy/recall/entity-recall, faithfulness, answer relevancy/similarity/correctness | +| **RAGAS** | โ€” | 15.2k โ˜… | v0.4.x; see ยง12.1 | + +**Recommendation for a small self-hosted team: promptfoo for local CI gating (genuinely offline, MIT, largest community); RAGAS or Braintrust autoevals for the RAG-specific scorer set; TruLens if you need to prove your judge agrees with your humans; Inspect AI if you need auditability. โš ๏ธ Phoenix's ELv2 license is a real constraint โ€” check it before embedding in a product.** + +## 12.4 The 2026 meta-shift: judge alignment over judge choice + +**Claim: the measurable 2026 change is that frameworks stopped shipping new metrics and started shipping **instrumentation to validate the judge** โ€” judgeโ†”human agreement reports, cross-model judge comparison, judge ensembles/juries, golden-set curation from production traces, few-shot judge optimization, and score-distribution calibration diagnostics โ€” all landed in TruLens 2.9โ€“2.12 between 2026-07-23 and 2026-08-06.** +โ†’ https://github.com/truera/trulens/releases ยท https://pypi.org/project/trulens/ +**Confidence: emerging, but the clearest signal of where practice moved** โ€” and consistent with the ยง9 finding that factuality metrics disagree with each other and misestimate system-level performance. + +## 12.5 Retrieval metrics practice โ€” unchanged + +**nDCG@10** as the headline (BEIR/BRIGHT/MTEB convention), **recall@k** for the retriever stage, **MRR@10** for MS MARCO-style single-answer settings. Provence, as a representative example, reports R@5 on NQ/HotpotQA, MRR@10 on MS MARCO, nDCG@10 on TREC DL'19, and mean nDCG@10 across 13 BEIR datasets. +โ†’ https://arxiv.org/abs/2501.16214. *established.* + +**For a RAG pipeline specifically: recall@k at the retriever and nDCG@10 after reranking are the two that matter, because generation quality is bounded by whether the evidence is in the window at all โ€” reinforced by the finding that retrieval is the main driver of attribution quality in both citation paradigms** (https://arxiv.org/abs/2509.21557). *established.* + +## 12.6 Leaderboards, Aug 2026 + +| Benchmark | Status | Source | +|---|---|---| +| **BEIR** | Still the zero-shot retrieval standard (13โ€“18 datasets, nDCG@10). *established* | โ€” | +| **MTEB** | Still the umbrella; **overfitting concerns now official** | https://huggingface.co/blog/rteb | +| **RTEB** โญ NEW | Maintainers' answer to overfitting. **Hybrid open + private datasets**; the openโ†”private gap directly measures overfitting. 20 languages, enterprise domains. Beta 2025-10-01. *established* | https://huggingface.co/blog/rteb | +| **BRIGHT** | Reasoning-intensive retrieval, 1,385 queries. Current leaders: **Mira-Reasoning-Retrieval 66.9 (2026-04-22), INF-X-Retriever 63.4, RakanEmbed4B 52.4, NeMo Retriever Agentic Retrieval 50.9 (open-weight)**. Baselines: **BM25 14.5** (30.4 with GPT-4 reasoning + rerank), GritLM-7B 21.0, instructor-xl 18.9, text-embedding-3-large 17.9, bge-large-en-v1.5 13.7. Most top submissions insert an **LLM reasoning step before retrieval**. *established* | https://brightbenchmark.github.io/ | +| BRIGHT โ€” โš ๏ธ caveat | Reproducibility audit found **undocumented implementation details (query-side BM25)**, corpus quality issues, and that **BM25Q gains are largely BRIGHT-specific rather than generalizable**. *contested* | https://arxiv.org/abs/2509.02558 (2025-09-02) | +| BRIGHT โ€” cost caveat | Across 12 BRIGHT tasks, LLM-based retrievers show **weak confidence signals and inconsistent reasoning-augmentation benefit across model families**. *emerging* | https://arxiv.org/abs/2604.03676 (2026-04-04) | +| **BRIGHT-Pro** โญ NEW | Expert-annotated expansion with multi-aspect gold evidence, evaluating retrievers under **both static and agentic protocols**; plus RTriever-Synth corpus and RTriever-4B. Core claim: retrievers must provide **"complementary evidence across iterative search and synthesis"**, not topical similarity; **"aspect-aware and agentic evaluation expose behaviors hidden by standard metrics."** *emerging* | https://arxiv.org/abs/2605.04018 (2026-05-06) | +| **MM-BRIGHT** โญ NEW | First multimodal reasoning-intensive retrieval benchmark, 29 technical domains. *emerging* | https://arxiv.org/abs/2601.09562 (2026-01-14) | +| **HAKARI-Bench** โญ NEW | Nano-dataset reconstruction across **43 languages**, 5 retrieval families; **correlates highly with official MTEB and BEIR rankings at a fraction of the compute.** Genuinely useful for a small team. *emerging* | https://arxiv.org/abs/2606.22778 (2026-06-22) | +| **RAGTruth** | Still the span/response hallucination standard; best open results now **~81.8 example-F1** (Qwen3.5-2B, Jul 2026). *established* | https://arxiv.org/abs/2607.00895 | +| **LLM-AggreFact** | Still the grounded-factuality standard. **Leaderboard appears frozen** โ€” newest datable entry Aug 2025; 2026 models (ThinknCheck 78.1) not listed. *contested as a live signal* | https://llm-aggrefact.github.io/ | +| **TRIVIA+** โญ NEW | RAG hallucination-detection with **the longest contexts in the literature** + four label sets simulating realistic label noise. ACL 2026. *emerging* | https://arxiv.org/abs/2605.11330 | +| **BenchPress** โญ NEW | Standardized context-compression eval suite. *emerging* | https://arxiv.org/abs/2510.20797 | +| **ragscale** โญ NEW | Compression-audit toolkit, 177k transitions; audits a compression paper with 3 readers in ~1 day. *emerging* | https://arxiv.org/abs/2606.21807 | +| **RAISE** โญ NEW | RAG hyperparameter/architecture-search benchmark; finding: **optimization is highly task-dependent โ€” methods strong on one dataset don't generalize.** *emerging* | https://arxiv.org/abs/2605.30029 | +| **RAGRouter-Bench** โญ NEW | First RAG-strategy routing benchmark (see ยง7) | https://arxiv.org/abs/2602.00296 | +| **BrowseComp-Plus** โญ | Retriever-disentangled agentic search eval, ACL 2026 (see ยง8) | https://aclanthology.org/2026.acl-long.1023/ | +| **OmniDocBench v1.6/1.7** | Document parsing (see ยง1) | https://github.com/opendatalab/OmniDocBench | + +## 12.7 Synthetic eval-set generation + +RAGAS ships test-data generation; TruLens added `GoldenSetGenerator` to curate eval datasets from production records (v2.9.0). A multi-agent framework generates diverse + privacy-masked synthetic RAG eval sets (https://arxiv.org/abs/2508.18929, 2025-08-26) but **does NOT validate against human-labeled evaluation sets.** + +**โš ๏ธ Gap flag: no 2026 primary source measures how well synthetic RAG eval sets agree with human-labeled ones. Treat synthetic eval sets as regression tripwires, not as ground truth for absolute quality claims.** Cross-reference ยง6.1: synthetic evals overstated the need for query augmentation by **60+ percentage points** in the one production system that measured it. + +--- + +# BIGGEST SHIFTS IN THE LAST 12 MONTHS + +**One paragraph:** The dominant story of Aug 2025 โ†’ Aug 2026 is that **the field turned self-critical and the small models won**. In parsing, sub-1B specialist VLMs (PaddleOCR-VL-1.6 at 96.34, GLM-OCR at 95.22, both 0.9B) decisively beat every frontier API on OmniDocBench and effectively saturated it, while the Western olmOCR lineage shipped no new model for ten months and Docling repositioned itself from parser to orchestration layer that swaps VLM backends. In embeddings, **Microsoft open-sourced Harrier under MIT and took the multilingual MTEB v2 crown at 74.3**, its 0.6B variant (69.0) now beating Qwen3-Embedding-0.6B by 4.7 points โ€” while MTEB's own maintainers launched **RTEB with private held-back test sets** because they concluded public leaderboards are overfit; the practical selection criterion shifted from score to **license**, since the two best sub-1B models (jina-v5, KaLM) are non-commercial. In verification, **a 4-bit 1B reasoning verifier (ThinknCheck, Apr 2026) beat the 7B 2024 SOTA on LLM-AggreFact**, making per-answer grounding checks cheap enough to always run โ€” even as span detectors were shown to **collapse from 0.69 to 0.17 span-F1 when the context is code or tool output** rather than prose. Compression took three independent hits in one year โ€” an ECIR 2026 systems study measuring **โ‰ค18% best-case speedup and net-negative outside a narrow window**, a demonstration that **fixed compression reverses 31% of model rankings and hides 80% of a reader upgrade**, and the discovery that hard compressors leave multi-hop answer paths incomplete in **34โ€“60% of bridge cases** โ€” leaving reranker-integrated sentence pruning (Provence / the MIT OpenProvence reimplementation) as the only variant whose economics survive. Memory's standard benchmark **LOCOMO went from canonical to discredited and saturated**, with the first independent controlled ablation (MemDelta, Jun 2026) finding Mem0 *loses* to plain cloud-RAG on LongMemEval-S and that **swapping only the embedding model moves accuracy more (+6.2pp) than most claimed memory-system gains**. On query planning, **HyDE was demoted from default to conditional** โ€” measurably โˆ’7.3% Recall@5 on numeric corpora, and needed for only 27.8% of real production queries versus the >90% synthetic evals implied โ€” with the consensus shifting to *escalate, don't pre-decide*: cheapest-first post-retrieval cascades rather than pre-retrieval classifiers, since **four ML approaches to pre-retrieval routing all failed**. Routing followed the same arc: **fixed hybrid BM25+dense with RRF beat rule-based retriever routing by +1.8 EM** on a local 7B, and where routing does pay (pipeline depth), a **TF-IDF + SVM classifier beat sentence embeddings** at 0.928 macro-F1. Agentic loops consolidated and productized โ€” **open-weight 4B deep-research agents now match Claude-4.5-Sonnet on GAIA-Text (71.3 vs 71.2)**, RL training cost collapsed ~13ร—, and Chroma shipped **Context-1, a 20B Apache-2.0 self-editing retrieval subagent** hitting 0.87โ€“0.96 on BrowseComp-Plus at 10ร— frontier inference speed โ€” but the year also produced the **first serious negative results on subagent fan-out** (41.8% of deep-agentic failures occur silently at the plannerโ†’subagent hand-off; plain semantic search beat deep agentic search 65.2% vs 46.2% at half the cost on repo-level code QA) and confirmed that **search volume correlates weakly with answer quality**, pushing termination criteria toward cumulative evidence sufficiency. Finally, evaluation tooling stopped shipping new metrics and started shipping **judge-validation instrumentation** (TruLens 2.9โ€“2.12: alignment reports, juries, cross-model comparison, golden-set generation), RAGAS took its sharpest criticism yet (**F1 0.570 at detecting corpus-evolution faults**), promptfoo was acquired by OpenAI, and **no new open judge model shipped at all** โ€” leaving Selene 1 Mini (Jan 2025, Apache 2.0) still the best permissive default. **The through-line for a self-hosted stack: the single highest-ROI investment is still the retriever and the reranker โ€” swapping BM25 for Qwen3-Embedding-8B moved the same agent +14.2 points on BrowseComp-Plus, and a cross-encoder is worth +17.2 pp MRR@3 โ€” while nearly every clever technique layered on top (HyDE, multi-query, compression, memory systems, subagent fan-out) was measured this year and found to be conditional at best, and a silent regression at worst.** + +--- + +## Source-quality caveats + +**Peer-reviewed anchors:** BrowseComp-Plus (ACL 2026 Main), Query Decomposition Exploration-Exploitation (EACL 2026), LLMLingua latency profiling (ECIR 2026), XProvence (ECIR 2026), Provence (ICLR 2025), MMTEB (ICLR 2025), TRIVIA+ (ACL 2026), OmniDocBench (CVPR 2025), MiniCheck (EMNLP 2024), HHEM (EMNLP 2025 Industry), PEA-CAE (IEEE SEGE 2026). + +**Unrefereed preprints carrying disproportionate weight โ€” treat as emerging:** the local-7B ablation (2606.21553, single author), RAGRouter-Bench baselines (2604.03455, in-distribution only), front-door SLM routing (2604.02367, neff=60), Coverage Illusion (2605.27220, single production system), MemDelta (2606.29914), ThinknCheck (2604.01652), plus most 2026 RL-search-agent papers. + +**Vendor self-reports (direction credible, magnitude favorable by construction):** all embedding/reranker model-card benchmarks, Chroma Context-1, Mem0, Zep, Letta, Vectara, Anthropic's multi-agent 90.2% figure, Jina's binary-quantization claim. + +**Flagged unreliable โ€” do not cite:** https://arxiv.org/abs/2607.27562 (HLE 87.3% / BrowseComp-ZH 85.3%) โ€” implausible and uncorroborated. + +**Withdrawn:** https://arxiv.org/abs/2601.04618 (REPAIR) โ€” withdrawn 2026-04-14 for missing institutional permissions. + +**Known gaps:** the MTEB/MMTEB and BrowseComp-Plus live leaderboards render client-side and could not be fetched (numbers above come from model cards and papers); the unstructured competitive positioning is unsourced; Flow Judge and JudgeLM current status unverified; RECOMP not re-verified; Cleanlab's detector-comparison numeric table unextractable; no 2026 benchmark exists for a sub-1B *generative* retrieval router; and the session WebSearch quota (200) was exhausted early, so vendor engineering blogs are under-sampled relative to arXiv. \ No newline at end of file diff --git a/Documentation/research/industry-evidence-2026.md b/Documentation/research/industry-evidence-2026.md new file mode 100644 index 00000000..36765aae --- /dev/null +++ b/Documentation/research/industry-evidence-2026.md @@ -0,0 +1,833 @@ +All four researchers returned. Here is the consolidated deliverable. + +--- + +# STATE OF THE ART IN AGENTIC RETRIEVAL โ€” RAW FINDINGS +**Compiled 2026-08-08. First-party engineering blogs, official docs, and named practitioners only.** + +**Method note:** ~180 first-party pages were fetched directly across four parallel research tracks plus a primary Anthropic/OpenAI sweep. Every URL below was actually fetched unless explicitly marked `[snippet-only]` or listed under GAPS. Dates are as printed on the page. The session's WebSearch quota (200) was exhausted; late-stage sources were reached by direct URL. Several notable pages 403 to automated fetch (`openai.com/index/*`, `help.openai.com`, `perplexity.ai/hub/*`) โ€” substitutes are noted. + +--- + +## SECTION 1 โ€” AGENTIC RAG vs CLASSIC PIPELINES: WHAT THE INDUSTRY CONVERGED ON + +### 1.1 The one clean vendor decision table + +**CLAIM:** LangChain ships an explicit three-way taxonomy with a tradeoff table. **2-Step RAG**: "the retrieval step is always executed before the generation step. This architecture is straightforward and predictable" โ€” Control High / Flexibility Low / Latency Fast; use case "FAQs, documentation bots." **Agentic RAG**: "an agentโ€ฆ reasons step-by-step and decides when and how to retrieve information during the interaction" โ€” Control Low / Flexibility High / Latency Variable; use case "Research assistants with multiple tools." **Hybrid RAG** in between. Explicit latency argument for the fixed pipeline: "Latency is generally more predictable in 2-Step RAG, as the maximum number of LLM calls is known and capped." +**SOURCE:** LangChain โ€” Retrieval (OSS docs) โ€” https://docs.langchain.com/oss/python/langchain/retrieval โ€” undated (LangChain 1.x) +**STATUS:** GA docs +*This is the single clearest published "when is the fixed pipeline still right" answer from any vendor.* + +**CLAIM:** LangChain's operational definition of agentic hinges on conditionality: "That choice is what makes the system agentic rather than a fixed retrieve-then-generate pipeline: retrieval runs only when the model requests it." Shipped patterns: query-or-respond gate, grade-documents-for-relevance, rewrite-question-and-re-retrieve (CRAG in LangGraph form). +**SOURCE:** LangChain โ€” Build a custom RAG agent with LangGraph โ€” https://docs.langchain.com/oss/python/langgraph/agentic-rag โ€” undated +**STATUS:** OSS pattern / shipped + +**CLAIM:** Microsoft made the choice a literal config enum. `TextSearchProvider.SearchTime` has exactly two values: `BeforeAIInvoke` (search runs prior to every model invocation = classic fixed pipeline, and it is the **default**) or on-demand via function calling (= agentic). Python bridges Semantic Kernel VectorStore collections into agent tools with `keyword_hybrid` and `semantic_hybrid`. +**SOURCE:** Microsoft โ€” Agent Framework, "RAG" โ€” https://learn.microsoft.com/en-us/agent-framework/agents/rag โ€” ms.date 2025-11-11, updated 2026-07-10 +**STATUS:** shipped (.NET/Python); Go "coming soon" + +**CLAIM:** OpenAI publishes **no** "agentic RAG vs naive RAG" framing at all. Retrieval is only ever a tool: agents get context via `instructions`, input messages, function tools for "on-demand context โ€” the LLM decides when it needs some data," and "Retrieval or web search" as "special tools" for "grounding the response in relevant contextual data." +**SOURCE:** OpenAI โ€” Agents SDK, Context management โ€” https://openai.github.io/openai-agents-python/context/ and Tools โ€” https://openai.github.io/openai-agents-python/tools/ โ€” undated +**STATUS:** GA โ€” *notable negative finding: OpenAI never validates the fixed-pipeline form in docs.* + +### 1.2 The framework vendors' own repositioning + +**CLAIM:** LlamaIndex: "naive RAG by itself isn't sufficient to meet enterprise needs" โ€” but names when it *is* sufficient: "search this handbook" and single-document summarization for human review. Adopt Agentic Document Workflows when workflows span multiple document types, enforce business rules, update systems of record, or run at high volume. +**SOURCE:** LlamaIndex / Jerry Liu โ€” "Beyond Chatbots: Adopting Agentic Document Workflows for Enterprises" โ€” https://www.llamaindex.ai/blog/beyond-chatbots-adopting-agentic-document-workflows-for-enterprises โ€” 2025-04-23 +**STATUS:** opinion-recommendation + product framing + +**CLAIM:** ADW is positioned as beyond *both* IDP and RAG: "a step beyond both traditional Intelligent Document Processing (IDP) and RAG paradigms, which are focused on small, isolated steps." Four-stage typed pipeline: Parse โ†’ Retrieve โ†’ Reason โ†’ Act. +**SOURCE:** LlamaIndex โ€” "Agentic Document Workflows: A Practical Guide" โ€” https://www.llamaindex.ai/blog/introducing-agentic-document-workflows โ€” 2025-01-09 +**STATUS:** OSS pattern + product + +**CLAIM โ€” MAJOR REVERSAL:** LlamaIndex now says its own category is receding. Verbatim: "if you equip agents with good filesystem tools, they can do dynamic search over document collections that outperforms naive semantic search," and RAG frameworks โ€” "the kind of thing LlamaIndex and LangChain (and some others) built โ€” aren't as central as they used to be." Retained moat is parsing: frontier VLMs "struggle with the long tail of accuracy parsing information-rich pages, like line charts, extremely dense tables (~hundreds of rows/columns in a single page), and handwritten forms." LlamaParse: 50+ formats, 500M+ pages processed. +**SOURCE:** LlamaIndex โ€” "LlamaIndex is more than a RAG Framework. It is Agentic Document Processing." โ€” https://www.llamaindex.ai/blog/llamaindex-is-more-than-a-rag-framework โ€” 2026-03-03 +**STATUS:** vendor repositioning + +**CLAIM:** Weaviate frames agentic retrieval as the default interaction layer for 2026 โ€” Query Agent GA, Transformation/Personalization Agents in preview, agents as "primary data interaction tools," 2026 focus on "shared memory systems enabling agents that learn iteratively." +**SOURCE:** Weaviate โ€” "Weaviate in 2025: Reliable Foundations for Agentic Systems" โ€” https://weaviate.io/blog/weaviate-in-2025 โ€” 2026-01-29 +**STATUS:** GA + preview mix / partly aspirational + +### 1.3 The infrastructure-side argument + +**CLAIM (strongest vendor statement of the thesis):** Vespa's CEO argues the fixed vector-search-then-generate pipeline is a category error. "Agents, in contrast, are not so clueless. And they certainly aren't lazy!" Agents should "string together many of these queries to reach its goal," needing proximity-based lexical search, quality-weighted semantic search, temporal filtering and aggregation, and *a menu of rank profiles to choose from* โ€” then run a sequenced fan-out: "First gaining an overview, then researching more specific topics, forming hypotheses, verifying important details in them." Three-stage maturity model: vector-only DB โ†’ hybrid + ML ranking โ†’ "search as code." +**SOURCE:** Vespa โ€” Jon Bratseth โ€” "Your agent wants to search like a 2010 quant" โ€” https://blog.vespa.ai/your-agent-wants-to-search-like-a-2010-quant/ โ€” 2026-07-07 +**STATUS:** vendor position piece (no numbers, no cost analysis) + +**CLAIM:** turbopuffer describes the workload shift concretely: "the LLM is very good at reasoning with the data. And so we're just the tool call," with agents issuing "an enormous amount of queries all at once" and "more concurrency than I've ever seen before." Cursor saw a 95% cost reduction post-migration; Notion does "a ridiculous amount of queries in every round trip." +**SOURCE:** turbopuffer โ€” "Retrieval After RAG: Hybrid Search, Agents, and Database Design" (Latent Space) โ€” https://turbopuffer.com/blog/podcast-latent-space โ€” 2026-03-12 +**STATUS:** shipped-in-production + +**CLAIM:** Pinecone re-architected for agent workloads specifically, characterizing them as "Millions of namespaces," "fewer than 100k vectors" per namespace, and "sporadic and bursty query patterns." Adaptive LSM-tree indexing on write, blob-storage persistence with on-demand fetch/cache on read โ†’ "response times of approximately 10ms" for small-namespace linear scans; >100M vectors at ">1000 QPS." +**SOURCE:** Pinecone โ€” "Optimizing Pinecone for agents (and more)" โ€” https://www.pinecone.io/blog/optimizing-pinecone/ โ€” 2025-03-17 +**STATUS:** shipped-in-production + +**CLAIM:** Databricks reports multi-step search agents beat single-step RAG with numbers: using their Knowledge Assistant as a tool inside a multi-step search agent beats RAG-as-a-tool by **over 30%** while *decreasing* time-to-completion by **8%**; "multi-step search agents are consistently more effective than single-step retrieval workflows." Their Instructed Retriever claims **~70%** over simplistic RAG, **~15%** over DIY reranking pipelines, and **35โ€“50%** higher recall on StaRK-Instruct. Architectural argument: RAG "loses context after initial retrieval"; the fix is propagating "complete system specifications โ€” from instructions to examples and index schema โ€” through every stage of the search pipeline." +**SOURCE:** Databricks โ€” "Instructed Retriever: Unlocking System-Level Reasoning in Search Agents" โ€” https://www.databricks.com/blog/instructed-retriever-unlocking-system-level-reasoning-search-agents โ€” 2026-01-06 +**STATUS:** GA (shipped in Agent Bricks Knowledge Assistant) + +**CLAIM:** Chroma's argument for why fixed pipelines fail specifically on multi-hop: single-pass retrieval "assumes that the information needed to answer a question can be retrieved in a single pass," whereas multi-hop needs "a chain of intermediate searches in which the output of one search informs the next." +**SOURCE:** Chroma โ€” "Chroma Context-1: Training a Self-Editing Search Agent" โ€” https://www.trychroma.com/research/context-1 โ€” 2026-03-26 +**STATUS:** research (Apache 2.0 weights, not a hosted product) + +### 1.4 The counter-case: when agentic loops lose + +**CLAIM:** Voyage published data that once first-stage retrieval is strong, putting an LLM in the loop as a ranker makes things *worse*: "Qwen 3 32B and Gemini 2.0 Flash actually degrade performance, with NDCG@10 dropping to 80.63% and 79.49%, respectively, from the baseline of 81.58%." LLM rerankers "cost 25-60x more than rerank-2.5" and are up to 48x slower. +**SOURCE:** Voyage AI โ€” "The Case Against LLMs as Rerankers" โ€” https://blog.voyageai.com/2025/10/22/the-case-against-llms-as-rerankers/ โ€” 2025-10-22 +**STATUS:** vendor research backing a GA product + +**CLAIM:** Elastic ships Agent Builder GA but simultaneously documents that its own reranker preview "is cost prohibitive for high query rates and low query latency requirements," capping CPU reranking at top-30. +**SOURCE:** Elastic โ€” Elastic Rerank docs โ€” https://www.elastic.co/docs/explore-analyze/machine-learning/nlp/ml-nlp-rerank โ€” undated +**STATUS:** preview + +**CLAIM:** Exa productizes the frontier explicitly rather than picking a side: Fast (sub-500ms) for voice/coding agents, Auto, Deep ("agentic search with multiple sequential queries," "a few seconds of latency"), Deep Max (11โ€“64s) for research. A deep-search MCP tool lets agents call the server **up to 10 times sequentially**. +**SOURCE:** Exa โ€” "Introducing Exa 2.1" โ€” https://exa.ai/blog/exa-api-2-1 โ€” 2025-11-24; "The World's Fastest Search API" โ€” https://exa.ai/blog/fastest-search-api โ€” 2025-07-29 +**STATUS:** GA + +**CLAIM (Anthropic's own brake):** "Only increasing complexity when needed"; agentic systems "trade latency and cost for better task performance." Workflows give "predictability and consistency for well-defined tasks"; agents are for "when flexibility and model-driven decision-making are needed at scale." Bottom line: "Success in the LLM space isn't about building the most sophisticated system. It's about building the *right* system for your needs." +**SOURCE:** Anthropic โ€” "Building Effective AI Agents" โ€” https://www.anthropic.com/engineering/building-effective-agents โ€” 2024-12-19 +**STATUS:** foundational guidance, never retracted + +--- + +## SECTION 2 โ€” ANTHROPIC & OPENAI PUBLISHED GUIDANCE + +### 2.1 Anthropic โ€” Contextual Retrieval (the pre-agentic baseline, still live) + +**CLAIM:** Contextual Retrieval prepends a 50โ€“100 token LLM-generated situating blurb to each chunk before embedding *and* before BM25 indexing. Prompt used verbatim: "Please give a short succinct context to situate this chunk within the overall document for the purposes of improving search retrieval of the chunk. Answer only with the succinct context and nothing else." Generated with Claude 3 Haiku. +**Numbers:** contextual embeddings alone โ†’ **35%** reduction in top-20 retrieval failure rate (5.7% โ†’ 3.7%); + contextual BM25 โ†’ **49%** (5.7% โ†’ 2.9%); + reranking โ†’ **67%** (5.7% โ†’ 1.9%). Pipeline shape: retrieve top-150 โ†’ rerank to top-20 โ†’ generate. Top-20 beat top-5 and top-10. One-time cost **$1.02 per million document tokens** with prompt caching. +**SOURCE:** Anthropic โ€” "Introducing Contextual Retrieval" โ€” https://www.anthropic.com/news/contextual-retrieval (also /engineering/contextual-retrieval) โ€” 2024-09-19 +**STATUS:** shipped technique / cookbook + +**CLAIM:** Anthropic's own "don't build RAG" threshold: "If your knowledge base is smaller than 200,000 tokens (about 500 pages of material), you can just include the entire knowledge base in the promptโ€ฆ with no need for RAG or similar methods," citing prompt caching as making this viable. +**SOURCE:** same as above โ€” 2024-09-19 +**STATUS:** shipped guidance + +### 2.2 Anthropic โ€” Context engineering (the pivot) + +**CLAIM:** Definition: context engineering is "the set of strategies for curating and maintaining the optimal set of tokens (information) during LLM inference, including all the other information that may land there outside of the prompts." Prompt engineering is the discrete-task subset; context engineering is iterative, running every turn. +**CLAIM:** Context is "a finite resource with diminishing marginal returns." Mechanism given: transformers create "nยฒ pairwise relationships for n tokens," and "models develop their attention patterns from training data distributions where shorter sequences are typically more common than longer ones." Outcome is a gradient, not a cliff โ€” "reduced precision for information retrieval and long-range reasoning." +**CLAIM โ€” the retrieval pivot:** a documented field shift from embedding-based pre-inference retrieval to **"just in time"** context. Verbatim: "Rather than pre-processing all relevant data up front, agents built with the 'just in time' approach maintain lightweight identifiers (file paths, stored queries, web links, etc.) and use these references to dynamically load data into context at runtime using tools." Claude Code "uses targeted queries, store results, and leverage Bash commands like head and tail to analyze large volumes of data without ever loading the full data objects into context." This enables "progressive disclosure โ€” โ€ฆ allows agents to incrementally discover relevant context through exploration." +**CLAIM โ€” the hybrid hedge:** "the most effective agents might employ a hybrid strategy, retrieving some data up front for speed, and pursuing further autonomous exploration at its discretion," recommended for "contexts with less dynamic content, such as legal or finance work." And: "as model capabilities improve, agentic design will trend towards letting intelligent models act intelligently, with progressively less human curation." Bottom line: "do the simplest thing that works." +**CLAIM โ€” sub-agent economics:** each subagent "explore[s] extensively, using tens of thousands of tokens or more, but return[s] only a condensed, distilled summary" โ€” "typically 1,000-2,000 tokens." +**CLAIM โ€” compaction craft:** "Start by maximizing recall to ensure your compaction prompt captures every relevant piece of information from the trace, then iterate to improve precision." +**CLAIM โ€” overarching principle:** "find the smallest set of high-signal tokens that maximize the likelihood of your desired outcome." +**SOURCE:** Anthropic โ€” "Effective context engineering for AI agents" โ€” https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents โ€” 2025-09-29 (Rajasekaran, Dixon, Ryan, Hadfield) +**STATUS:** guidance backed by shipped features + +**CLAIM (the nuanced code-search position, often mis-quoted):** "Semantic search is usually faster than agentic search, but less accurate, more difficult to maintain, and less transparentโ€ฆ we suggest starting with agentic search, and only adding semantic search if you need faster results." +**SOURCE:** same post โ€” 2025-09-29 + +### 2.3 Anthropic โ€” tool design for retrieval + +**CLAIM:** Consolidate rather than proliferate: "Instead of implementing a `read_logs` tool, consider implementing a `search_logs` tool which only returns relevant log lines and some surrounding context." Tools "should take care to return only high signal information back to agents," with "pagination, range selection, filtering, and/or truncation with sensible default parameter values." Claude Code caps tool responses at **25,000 tokens** by default. +**CLAIM:** Natural-language identifiers beat UUIDs โ€” "merely resolving arbitrary alphanumeric UUIDs to more semantically meaningful and interpretable languageโ€ฆ significantly improves Claude's precision in retrieval tasks by reducing hallucinations." +**CLAIM:** `response_format` enum (concise vs detailed) โ€” Slack example: concise used ~1/3 the tokens (72 vs 206). +**CLAIM:** Namespace by service and resource (`asana_search`, `asana_projects_search`). And: "If a human engineer can't definitively say which tool should be used in a given situation, an AI agent can't be expected to do better." +**SOURCE:** Anthropic โ€” "Writing effective tools for AI agentsโ€”using AI agents" โ€” https://www.anthropic.com/engineering/writing-tools-for-agents โ€” 2025-09-11 +**STATUS:** guidance + +### 2.4 Anthropic โ€” shipped retrieval-adjacent API surface (status matters here) + +| Feature | Mechanism | Status (Aug 2026) | Key numbers | +|---|---|---|---| +| **Web search tool** | Server-side loop; "The API runs the searchesโ€ฆ This process can repeat multiple times throughout a single request." `max_uses` hard cap. Citations always on. | **GA** | **$10 per 1,000 searches**. "Simple factual queries typically use 1โ€“3 searches; comparative or multientity research can use 10 or more." `cited_text` (โ‰ค150 chars), `title`, `url` don't count toward tokens. | +| **Web search dynamic filtering** (`web_search_20260209`+) | Claude writes and runs code inside code execution that **filters results before they enter context**; `allowed_callers` defaults to `["code_execution_20260120"]` | **GA** | No extra charge for the code execution calls | +| **Web search response inclusion** (`web_search_20260318`+) | `response_inclusion: "excluded"` drops raw search blocks from the API response for agentic workflows | **GA** | Reduces output token cost | +| **Web fetch tool** | Claude "is not allowed to dynamically construct URLs" โ€” can only fetch URLs already in conversation or from prior search results (`url_not_in_prior_context` error) | **GA** | Free beyond tokens. 10 kB page โ‰ˆ 2,500 tokens; 500 kB PDF โ‰ˆ 125,000 tokens | +| **Search result content blocks** | `{"type":"search_result", source, title, content[], citations:{enabled}}` โ€” from tool calls or top-level user content; gives custom RAG the same citation quality as web search | **GA, no beta header**, all active models except Haiku 3 | โ€” | +| **Citations** | Sentence-chunked for PDFs/plain text; custom content blocks used as-is. Returns `char_location` / `page_location` / `content_block_location` | **GA**, all active models | `cited_text` **does not count toward output tokens**, and not toward input tokens when passed back. Anthropic: "the citations feature is significantly more likely to cite the most relevant quotes from documents than purely prompt-based approaches" | +| **Memory tool** (`memory_20250818`) | Client-side file CRUD under `/memories`: view, create, str_replace, insert, delete, rename. Framed as "just-in-time context retrieval." | **GA, no beta header**, all Claude 4+ | API auto-injects: "ALWAYS VIEW YOUR MEMORY DIRECTORY BEFORE DOING ANYTHING ELSEโ€ฆ ASSUME INTERRUPTION" | +| **Context editing** (`clear_tool_uses_20250919`, `clear_thinking_20251015`) | Server-side clearing of oldest tool results; `trigger` default 100k input tokens, `keep` default 3 tool uses, `clear_at_least`, `exclude_tools`, `clear_tool_inputs` | **Beta** (`context-management-2025-06-27`) | Launch numbers: **39%** improvement on internal agentic search eval (memory+editing), **29%** editing alone, **84%** token reduction in a 100-turn web search eval | +| **Compaction** (`compact_20260112`) | Server-side whole-conversation summarization; default trigger **150,000** input tokens, min 50,000; API drops all blocks before the compaction block | **Beta** (`compact-2026-01-12`) | Anthropic calls it "the recommended strategy for managing context in long-running conversations and agentic workflows" | +| **Tool Search Tool** (`defer_loading: true`) | Retrieval applied to tool *definitions* | **Beta** (`advanced-tool-use-2025-11-20`) | ~77K โ†’ ~8.7K tokens, "**85% reduction**โ€ฆ preserving 95% of context window." MCP eval accuracy: Opus 4 **49% โ†’ 74%**; Opus 4.5 **79.5% โ†’ 88.1%** | +| **Programmatic Tool Calling** | Claude writes Python that calls tools; intermediate results never enter context | **Beta** | **43,588 โ†’ 27,297 tokens (37%)** on complex research; knowledge retrieval **25.6% โ†’ 28.5%**; GIA **46.5% โ†’ 51.2%** | +| **Tool Use Examples** (`input_examples`) | Concrete usage patterns beyond JSON Schema | **Beta** | accuracy **72% โ†’ 90%** on complex parameter handling | +| **Agent Skills** | Three-level progressive disclosure: L1 name+description in system prompt (~100 tokens/skill), L2 SKILL.md body on trigger (<5k tokens), L3 bundled files at zero token cost until read | **GA** (API beta header `skills-2025-10-02`) | "the amount of context that can be bundled into a skill is effectively unbounded" | + +**SOURCES:** https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-search-tool ยท /web-fetch-tool ยท /memory-tool ยท https://platform.claude.com/docs/en/build-with-claude/search-results ยท /citations ยท /context-editing ยท /compaction ยท https://platform.claude.com/docs/en/agents-and-tools/agent-skills/overview ยท https://www.anthropic.com/engineering/advanced-tool-use (2025-11-24) ยท https://claude.com/blog/context-management (2025-09-29) ยท https://claude.com/blog/skills (2025-10-16) โ€” all accessed 2026-08-08 + +### 2.5 Anthropic โ€” MCP and the tool-token tax + +**CLAIM:** "58 tools consuming approximately 55K tokens before the conversation even starts"; agents with thousands of tools must process "hundreds of thousands of tokens before reading a request." Second problem: intermediate results pass through context twice โ€” a two-hour Google Drive transcript into Salesforce "potentially adding 50,000 tokens." +**CLAIM โ€” the fix:** present MCP servers as code APIs on a filesystem (`./servers/google-drive/getDocument.ts`); "models are great at navigating filesystems. Presenting tools as code on a filesystem allows models to read tool definitions on-demand, rather than reading them all up-front." Result: "**150,000 tokens to 2,000 tokens โ€” a time and cost saving of 98.7%**." Filtering a 10,000-row spreadsheet "to five rows instead of 10,000" in the execution environment. `search_tools` supports detail levels ("name only, name and description, or the full definition with schemas"). +**SOURCE:** Anthropic โ€” "Code execution with MCP: building more efficient AI agents" โ€” https://www.anthropic.com/engineering/code-execution-with-mcp โ€” 2025-11-04 (Adam Jones, Conor Kelly) +**STATUS:** OSS pattern / recommendation + +**PRIORITY FLAG:** Cloudflare published the same thesis **39 days earlier** with a different rationale โ€” a *training-data* argument: LLMs have seen enormous amounts of real TypeScript but comparatively little synthetic tool-call syntax. "If you present an LLM with too many tools, or overly complex tools, it may struggle to choose the right one"; "the output of each tool call must feed into the LLM's neural network, just to be copied over to the inputs of the next call, wasting time, energy, and tokens." Cloudflare does not cite Anthropic or prior work. +**SOURCE:** Cloudflare โ€” "Code Mode: the better way to use MCP" โ€” https://blog.cloudflare.com/code-mode/ โ€” 2025-09-26 (Kenton Varda, Sunil Pai) +**STATUS:** shipped (Workers/Agents SDK) + +**CLAIM:** MCP tool search is now **default-on** in Claude Code: "MCP tool definitions are deferred by default and loaded on demand via tool search, so only tool names consume context until Claude uses a specific tool." +**SOURCE:** Anthropic โ€” "How Claude Code works" โ€” https://code.claude.com/docs/en/how-claude-code-works +**STATUS:** shipped default + +**CLAIM (quiet qualification of MCP, from MCP's own author):** "CLI tools are the most context-efficient way to interact with external services" โ€” recommends installing `gh` rather than going through the GitHub API. +**SOURCE:** Anthropic โ€” Claude Code best practices โ€” https://code.claude.com/docs/en/best-practices +**STATUS:** shipped guidance + +**CLAIM (MCP spec):** MCP models retrieval two ways โ€” **resources** (application-driven, URI-addressed, with `resources/list`, `resources/read`, URI templates, optional `subscribe`/`listChanged`, and annotations carrying `audience`/`priority`/`lastModified`) and **tools** (model-driven). The spec mandates neither for search; in practice the ecosystem converged on tools. Notably, **OpenAI's deep research API imposes the tool contract**: remote MCP servers must implement "A `search` tool that takes a query and returns search results. A `fetch` tool that takes an id from the search results and returns the corresponding document." +**SOURCES:** MCP spec โ€” https://modelcontextprotocol.io/specification/2025-06-18/server/resources โ€” rev 2025-06-18 ยท OpenAI โ€” https://developers.openai.com/api/docs/guides/deep-research +**STATUS:** standard / GA + +**CLAIM:** Anthropic's Managed Agents virtualizes sessions (append-only event logs), harnesses (orchestration loops), and sandboxes. Retrieval note: all tools use a standardized `execute(name, input) โ†’ string` interface covering custom tools, MCP servers, and Anthropic's own. Context is retrievable by positional slice via `getEvents()`; "context can be an object in a REPL that the LLM programmatically accesses by writing code to filter or slice it." OAuth tokens live in a vault outside the sandbox behind an MCP proxy so "the harness is never made aware of any credentials." +**SOURCE:** Anthropic โ€” "Scaling Managed Agents: Decoupling the brain from the hands" โ€” https://www.anthropic.com/engineering/managed-agents โ€” 2026-04-08 +**STATUS:** hosted service, available + +### 2.6 OpenAI โ€” retrieval guidance and hosted retrieval defaults + +**CLAIM โ€” `file_search` / vector store defaults (the most-copied numbers in the industry):** `max_chunk_size_tokens` = **800**, `chunk_overlap_tokens` = **400** (the "auto" strategy uses exactly these). Configurable range 100โ€“4,096; overlap must not exceed half the chunk size. Embedding model referenced: `text-embedding-3-small`. +**CLAIM:** Shipped query-understanding and ranking knobs: `rewrite_query=true` (rewritten form returned in `search_query`), `ranking_options` with `ranker` (`auto` / `default-2024-08-21`) and `score_threshold` (0.0โ€“1.0), hybrid tuning via `embedding_weight` and `text_weight`, attribute filtering (max 16 keys, 256 chars, `eq/ne/gt/gte/lt/lte/in/nin` + `and`/`or`). Standalone search endpoint returns 10 by default, max 50. Limits: 512 MB and 5M tokens per file. Storage: first 1 GB free, then **$0.10/GB/day**. +**SOURCE:** OpenAI โ€” Retrieval guide โ€” https://developers.openai.com/api/docs/guides/retrieval โ€” accessed 2026-08-08; File search tool โ€” https://developers.openai.com/api/docs/guides/tools-file-search; Vector stores API ref โ€” https://developers.openai.com/api/docs/api-reference/vector-stores/create +**STATUS:** GA +*Note: the docs advertise "semantic and keyword search" but publish no detail on the hybrid mechanics or reranker internals.* + +**CLAIM โ€” `web_search` as an orchestration spectrum:** three explicitly named tiers โ€” "non-reasoning web search" (quick lookup), "agentic search with reasoning models," and "deep researchโ€ฆ using hundreds of sources." `search_context_size` low/medium/high tunes how much result content enters context but "does not set an exact token count or guarantee a specific number of sources." Web search context is capped at **128k even when the model's context window is larger**. Up to **100** `allowed_domains` or `blocked_domains`. A `sources` field returns every URL consulted โ€” "typically exceeds the number of inline citations." Inline citations "must be made clearly visible and clickable." +**SOURCE:** OpenAI โ€” Web search tool guide โ€” https://developers.openai.com/api/docs/guides/tools-web-search โ€” accessed 2026-08-08 +**STATUS:** GA + +**CLAIM โ€” GPT-5-era prompting guidance on bounding agentic search.** OpenAI ships prompt patterns for calibrating "agentic eagerness." To reduce search: lower `reasoning_effort` ("reduces exploration depth but improves efficiency and latency"); set early-stop criteria ("You can name exact content to change. Top hits converge (~70%) on one area/path."); impose a fixed tool budget ("Usually, this means an absolute maximum of 2 tool calls. If you think you need more time to investigate, update the user."); and give an escape hatch ("even if it might not be fully correct"). The `` block: "Start broad, then fan out to focused subqueries. In parallel, launch varied queries; read top hits per queryโ€ฆ Avoid over searching for context." / "Batch search โ†’ minimal plan โ†’ complete task. Search again only if validation fails or new unknowns appear. **Prefer acting over more searching.**" +**SOURCE:** OpenAI Cookbook โ€” GPT-5 prompting guide โ€” https://developers.openai.com/cookbook/examples/gpt-5/gpt-5_prompting_guide โ€” no date printed on page +**STATUS:** GA guidance +*This is the closest thing OpenAI has to Anthropic's "start wide then narrow" โ€” and it points the opposite direction: bias toward less search.* + +**CLAIM:** OpenAI Agents SDK compaction is a session wrapper that **does not summarize**: `OpenAIResponsesCompactionSession` "clears and rewrites history rather than summarizing it." Other levers are `SessionSettings(limit=N)`, `session_input_callback`, `pop_item()`. Session backends: SQLite, AsyncSQLite, OpenAIConversations, Redis, SQLAlchemy, MongoDB, Dapr, AdvancedSQLite, EncryptedSession. +**SOURCE:** OpenAI โ€” Agents SDK Sessions โ€” https://openai.github.io/openai-agents-python/sessions/ +**STATUS:** shipped โ€” *the weakest compression story of the four majors: no summarization node, no tool-result pruning primitive.* + +**CLAIM โ€” ChatGPT company knowledge:** "powered by a version of GPT-5 that's trained to look across multiple sources to give more comprehensive and accurate answers." Streams intermediate looking-at steps; returns the specific snippets used with citations; respects existing company permissions. Apps must expose **File Search plus `search` and `fetch` actions** to be eligible. Connectors were renamed "apps" on 2025-12-17. Connectors are "access connectors, which fetch content when a user asks a question, built using MCP." +**SOURCE:** OpenAI Help Center โ€” "Company knowledge in ChatGPT (Business, Enterprise, and Edu)" โ€” https://help.openai.com/en/articles/12628342-company-knowledge-in-chatgpt-business-enterprise-and-edu โ€” `[snippet-only; 403 to fetch]` +**STATUS:** shipped-in-production + +--- + +## SECTION 3 โ€” WHAT PRODUCTION SYSTEMS USE AT EACH STAGE + +### 3.1 Query understanding / decomposition / routing + +**CLAIM:** Weaviate Query Agent went **GA 2025-09-17** after ~6 months of preview. Shipped feature list *is* a query-understanding pipeline as a service: query planning and cross-collection routing, decomposition into concurrent searches, dynamic schema-valid filter construction, query expansion, reranking/aggregation, answer citation. Two modes: Ask (with generation) and Search (retrieval-only). +**SOURCE:** Weaviate โ€” "Accelerating Data Workflows with Query Agent, now GA" โ€” https://weaviate.io/blog/query-agent-generally-available โ€” 2025-09-17 +**STATUS:** GA + +**CLAIM:** Weaviate published head-to-head numbers for agentic Search Mode vs plain hybrid (Arctic 2.0 dense + BM25 via RRF): overall **+17% Success@1, +11% Recall@5**. Natural Questions Success@1 0.43 โ†’ 0.52, Recall@5 0.70 โ†’ 0.81; EnronQA Success@1 0.56 โ†’ 0.74; **BRIGHT Biology Success@1 0.13 โ†’ 0.44**, Recall@5 0.11 โ†’ 0.35. Credited to "query expansion, query decomposition, schema introspection, and reranking." **No latency reported.** +**SOURCE:** Weaviate โ€” "Search Mode Benchmarking" โ€” https://weaviate.io/blog/search-mode-benchmarking โ€” 2025-09-23 +**STATUS:** GA benchmark + +**CLAIM:** Elastic Agent Builder reached **GA** (Cloud Serverless and 9.3) with query planning as a first-class capability: "generate[s] optimized hybrid, semantic, and structured queries" from natural language, automatic index selection, MCP exposure. Pitch is consolidation โ€” removes the need for "separate data stores, vector databases, RAG pipelines, search layers, query translators, and tool orchestrators." +**SOURCE:** Elastic โ€” "Agent Builder now GA: Ship context-driven agents in minutes" โ€” https://www.elastic.co/search-labs/blog/agent-builder-elastic-ga โ€” 2026-01-22 +**STATUS:** GA + +**CLAIM:** Azure AI Search agentic retrieval is a four-stage pipeline: (1) app calls a knowledge base with query + conversation history; (2) **query planning** โ€” an LLM decomposes into focused subqueries; (3) **query execution** โ€” "All subqueries run simultaneously," each keyword/vector/hybrid, each **semantically reranked (L2) independently**, references retained for citation; (4) synthesis into a unified response plus an execution activity log. Reasoning effort is an explicit knob: `minimal` **skips query planning entirely**; `low` (default) and `medium` invoke the planner. +**SOURCE:** Microsoft โ€” "Agentic retrieval in Azure AI Search" โ€” https://learn.microsoft.com/en-us/azure/search/search-agentic-retrieval-concept โ€” ms.date 2026-06-02, updated 2026-07-02 +**STATUS:** **GA in the 2026-04-01 REST API**; Azure portal and Foundry portal remain **preview-only** + +**CLAIM โ€” Microsoft's published cost model for agentic retrieval:** worked example assumes **3 subqueries per plan**, **50 chunks reranked per subquery**, 500 tokens/chunk, 2,000 input tokens of chat history, 350-token output plan. For 2,000 retrievals: $3.30 Azure AI Search reranking + $1.02 Azure OpenAI planning = **$4.32**. Billing shifts from per-query (classic) to **per-token** (agentic). Stated plainly: "Agentic retrieval adds latency compared to a single-query pipeline." Cost guidance: "Lower the reasoning effortโ€ฆ Reduce the number of knowledge sources (indexes); consolidating content can lower **fan-out**." +**SOURCE:** same as above +**STATUS:** GA + +**CLAIM:** Elastic separately published an agentic-search-plus-autotuning reference architecture: an LLM agent fills search templates (V1โ€“V4) from natural language; an XGBoost Learn-to-Rank model with **48 features** (property attributes + engagement signals) reranks, retrained from logged interactions. +**SOURCE:** Elastic โ€” "Agentic search: autotuning relevance in Elasticsearch" โ€” https://www.elastic.co/search-labs/blog/agentic-search-relevance-autotuning-elasticsearch โ€” 2025-11-19 +**STATUS:** reference architecture, not a SKU + +**CLAIM:** Jason Liu argues query understanding should surface *facets and metadata*, not just top-k chunks โ€” "Agent Peripheral Vision: Providing agents with structured metadata about the broader information space beyond just the top-k results." Client numbers cited (unaudited): 90% reduction in clarification questions, 75% reduction in expert escalations, 4x improvement in resolution times. +**SOURCE:** Jason Liu โ€” https://jxnl.co/writing/2025/08/27/facets-context-engineering/ โ€” 2025-08-27 +**STATUS:** practitioner field report + +**NEGATIVE FINDING:** No first-party vendor page was found shipping **HyDE** as a named GA feature. Vendors ship query expansion, decomposition, rewriting, and planning โ€” not HyDE by name. Flagged as under-searched rather than proven absent. + +### 3.2 Hybrid retrieval โ€” and the RRF-vs-weighted split + +**CLAIM:** Across every embedding model Vespa tested, hybrid beat semantic-only: "Every single model scored higher with hybrid retrieval than semantic-only. On average, the best hybrid method beats semantic-only by **3-5 percentage points**." Storage for 100M ร— 768-dim: FP32 307 GB / FP16 154 GB / INT8 77 GB / binary 9.6 GB (32x). "Vespa can do ~1 billion hamming distance calculations per second, roughly 7x more than prenormalized angular distance." INT8 on CPU "2.7-3.4x faster while keeping 94-98% of the quality"; INT8 on GPU is *4-5x slower* than FP32. +**SOURCE:** Vespa โ€” "Embedding Tradeoffs, Quantified" โ€” https://blog.vespa.ai/embedding-tradeoffs-quantified/ โ€” 2026-01-14 +**STATUS:** benchmark on GA features + +**CLAIM โ€” vendors disagree on fusion.** Weaviate moved its default **off RRF**: `relativeScoreFusion` (min-max normalize, then weighted sum) is default from v1.24; `rankedFusion` (pure RRF) was default in โ‰คv1.23. Docs: "it retains more information from the original searches than `rankedFusion`, which only retains the rankings." Server default `alpha = 0.75` (vector-leaning). +**SOURCE:** Weaviate โ€” Hybrid search concepts โ€” https://docs.weaviate.io/weaviate/concepts/search/hybrid-search; "Hybrid Search Explained" โ€” https://weaviate.io/blog/hybrid-search-explained โ€” 2025-01-27 +**STATUS:** GA + +**CLAIM (opposite):** Qdrant calls RRF "the de facto standard in the field" and documents only RRF. +**SOURCE:** Qdrant โ€” "Hybrid Search with Qdrant's Query API" โ€” https://qdrant.tech/articles/hybrid-search/ โ€” 2024-07-25 +**STATUS:** GA + +**CLAIM (hedge):** MongoDB GA'd both โ€” `$rankFusion` (RRF over ranks) on 8.0+, `$scoreFusion` (normalized weighted average) on 8.2+. Customer quote: "This has improved the context retrieval accuracy for our Eddy AI chatbot by 30%" (Kovai.co). +**SOURCE:** MongoDB โ€” https://www.mongodb.com/company/blog/product-release-announcements/boost-search-relevance-mongodb-atlas-native-hybrid-search โ€” 2025-06-25, updated 2026-06-30 +**STATUS:** GA + +**CLAIM (third position โ€” skip fusion, rerank instead):** Pinecone's cascading retrieval yields "up to 48% better performance โ€” and 24% better, on average" over dense alone, and "**8% better than score fusion**" on BEIR. Their learned sparse model `pinecone-sparse-english-v0` gives "Up to 44% (average 23%) better NDCG@10" on TREC DL and "up to 24% (8% on average)" on BEIR vs BM25. +**SOURCE:** Pinecone โ€” "Introducing cascading retrieval" โ€” https://www.pinecone.io/blog/cascading-retrieval/ โ€” 2024-12-02 +**STATUS:** GA + +**CLAIM:** Pinecone positions SPLADE as not production-ready โ€” word-piece SPLADE embeddings are "still new and highly experimental." Sparse storage is "1000x smaller (and cheaper)" than dense at 100M scale. +**SOURCE:** Pinecone โ€” "Don't be dense: Launching sparse indexes in Pinecone" โ€” https://www.pinecone.io/learn/sparse-retrieval/ โ€” 2025-03-05 +**STATUS:** preview at time of writing + +**CLAIM:** Elastic's default semantic path is **sparse-neural, not dense**: `semantic_text` with no inference endpoint specified uses ELSER by default. Hybrid retrievers support "linear/generic rescoring alongside Reciprocal Rank Fusion (RRF)." BBQ GA'd with "up to 5x faster queries and 3.9x higher throughput" vs OpenSearch FAISS. +**SOURCE:** Elastic โ€” "What's new in Elastic 9.0 / 8.18" โ€” https://www.elastic.co/blog/whats-new-elastic-search-9-0-0 โ€” 2025-04-14 +**STATUS:** GA + +**CLAIM โ€” BM25 is being re-invested in *because of* agents:** turbopuffer's FTS v2 claims "up to 20x better full-text search performance" via a 10x smaller on-disk index plus MAXSCORE dynamic pruning. On ~5M Wikipedia docs, k=100: "lord of the rings" 75ms โ†’ 6ms; multi-term 174ms โ†’ 20ms. Rationale verbatim: "full-text search is equally important for recall and performance in agent-initiated queries," since agents write longer queries than humans. +**SOURCE:** turbopuffer โ€” "FTS v2: up to 20x faster full-text search" โ€” https://turbopuffer.com/blog/fts-v2 โ€” 2026-02-03 +**STATUS:** shipped-in-production + +**CLAIM โ€” a reference agent-memory retrieval stack with real numbers:** RRF over BM25 + Jina v5 dense (over-fetching 80 candidates per leg, `rank_constant=30`), then a Jina v2 cross-encoder on the merged pool; a single unified `recall_memory` tool spanning three memory indices; pre-recall on verbatim user messages to bypass LLM paraphrasing; `refresh=True` on episodic writes. Over 168 questions: **R@10 = 0.89** (0.85โ€“0.893 across four runs), R@5 0.75; semantic facts R@10 โ‰ˆ 0.81, episodic 0.98, procedural 1.0; **zero cross-tenant leaks** under Elasticsearch DLS per-user API keys. +**SOURCE:** Elastic Search Labs โ€” "Agent memory on Elasticsearch: hybrid retrieval and DLS" โ€” https://www.elastic.co/search-labs/blog/agent-memory-elasticsearch โ€” 2026-06-16 +**STATUS:** reference implementation with documented benchmark + +### 3.3 Late interaction / ColBERT / ColPali / MUVERA โ€” production status is vendor-dependent + +| Vendor | Position | Status | Numbers | +|---|---|---|---| +| **Vespa** | Verbatim FAQ: "Is Long-ColBERT in Vespa ready for production? **Yes.**" `tensor(context{}, token{}, v[16])`, 16 bytes/token vector | shipped-in-production | Reranks top-10 "below 50ms," significant nDCG@10 gain over BM25 | +| **Vespa (ColPali)** | 1 PDF page = 1,030 vectors ร— 128 dims. Binary quantization to 16-dim int8 โ†’ **32x storage reduction**, hamming MaxSim "~3.5x faster than float dot product," "~200M 128-bit hamming distances per second per CPU core" | shipped | DocVQA nDCG@5: float-float 52.4, binary-binary 49.5, **binary + float rerank 51.6** | +| **Weaviate** | MUVERA **GA in 1.31+**. Formula `repetitions * 2^ksim * dprojections` (defaults 4/16/10). Multi-vectors support PQ/BQ/RQ/SQ | **GA** | Memory โˆ’~70%; 12GB โ†’ <1GB; import 20+ min โ†’ 3โ€“6 min (110k objects). **Candid recall cliff: needs ef>512 for 80%+, ef 2048 for >90%**, "decreasing the query throughput" | +| **Qdrant** | MUVERA is **in FastEmbed 0.7.2+, not in the engine**. Late interaction recommended **only as a reranker, never first-stage** (store multivectors with HNSW `m=0`) | client library only | Full multivector 1.27s โ†’ MUVERA-only 0.15s (~8x) โ†’ MUVERA+rerank 0.18s. NDCG@10 0.347 โ†’ 0.343; MUVERA alone recovers only ~70% of quality | +| **Elastic** | "ColPali and ColBERT now supported with MaxSim" | GA per release blog | โ€” | +| **Pinecone** | **No native multi-vector support.** Own research stores ConstBERT embeddings "as metadata." Verdict: "end-to-end ColBERT is notably slow, even when using the official PLAID engine" (hundreds of ms); as a reranker "MaxSim computation for hundreds of documents takes only a few milliseconds." Cascade: ~1000 โ†’ 100 (multi-vector) โ†’ 10 (cross-encoder) | research-only | MSMARCO nDCG@10: ColBERT e2e 74.6, ConstBERT e2e 73.1, ConstBERT-rerank 74.4; BEIR avg ColBERT 48.8 vs ConstBERT-rerank 50.2 | +| **Vectara** | Deliberately rejecting it โ€” Boomerang successor is "a single-vector architecture intended to work with conventional vector-search infrastructure without the vector-volume increase associated with patch-level multi-vector retrieval," naming ColPali/VLM2Vec as what it avoids | aspirational / in-development | โ€” | + +**SOURCES:** Vespa โ€” https://blog.vespa.ai/announcing-long-context-colbert-in-vespa/ (2024-03-01, Bergum) ยท https://blog.vespa.ai/scaling-colpali-to-billions/ (2024-09-20, Bergum) ยท Weaviate โ€” https://weaviate.io/blog/muvera (2025-06-05), https://weaviate.io/blog/weaviate-1-31-release (2025-06-03), https://docs.weaviate.io/weaviate/configuration/compression/multi-vectors ยท Qdrant โ€” https://qdrant.tech/articles/muvera-embeddings/ (2025-09-05), https://qdrant.tech/articles/hybrid-search/ (2024-07-25) ยท Elastic โ€” https://www.elastic.co/blog/whats-new-elastic-search-9-0-0 (2025-04-14) ยท Pinecone โ€” https://www.pinecone.io/blog/cascading-retrieval-with-multi-vector-representations/ (2025-05-28) ยท Vectara โ€” https://www.vectara.com/blog/moving-beyond-text-conversion-the-future-of-enterprise-search (2026-08-05) + +**CLAIM:** Ben Claviรฉ / answer.ai published the token-pooling result that made ColBERT storage tractable ("considerable memory & disk footprint reduction"), plus `answerai-colbert-small` and the `rerankers` library unifying cross-encoder / ColBERT / LLM ranking under one API. +**SOURCES:** https://www.answer.ai/posts/colbert-pooling.html (2024-06-27) ยท https://www.answer.ai/posts/2024-08-13-small-but-mighty-colbert.html (2024-08-13) ยท https://www.answer.ai/posts/2024-09-16-rerankers.html (2024-09-16) +**STATUS:** open-source research/tooling + +### 3.4 Reranking + +**CLAIM:** Elastic Rerank is a **184M-param DeBERTa-v3 cross-encoder** (86M backbone + 98M embedding layer), distilled from a bi-encoder+cross-encoder ensemble over ~3M queries incl. ~180k synthetic pairs. BEIR nDCG@10: BM25 0.426 โ†’ **0.565**, vs Cohere v3 0.529 and bge-reranker-v2-gemma (2B) 0.568 โ€” "an average improvement of 39% across the full suite." Per-dataset: NQ 90%, MS MARCO 85%, Climate-FEVER 80%, FiQA-2018 76%. +**CRITICAL STATUS CAVEAT:** still **technical preview** as of current docs; needs a 4GB ML node standalone / "at minimum an 8GB ML node" with ELSER; Elastic's own warning: "the preview version is cost prohibitive for high query rates and low query latency requirements"; "We would recommend shallow reranking for CPU inference: **no more than top-30 results**." +**SOURCES:** Elastic โ€” https://www.elastic.co/search-labs/blog/elastic-semantic-reranker-part-2 (2024-11-25) ยท https://www.elastic.co/search-labs/blog/elastic-rerank-model-introduction (2024-12-10) ยท docs https://www.elastic.co/docs/explore-analyze/machine-learning/nlp/ml-nlp-rerank +**STATUS:** preview + +**CLAIM:** Voyage rerank-2.5 / rerank-2.5-lite are **instruction-following** with 32K context ("8x that of Cohere Rerank v3.5"), improving retrieval accuracy "by 7.94% and 7.16% over Cohere Rerank v3.5" across 93 datasets, "12.70% and 10.36%" on MAIR. +**SOURCE:** Voyage AI โ€” https://blog.voyageai.com/2025/08/11/rerank-2-5/ โ€” 2025-08-11 +**STATUS:** GA + +**CLAIM (the sharpest anti-LLM-reranker data):** NDCG@10 โ€” rerank-2.5 **84.32%**, rerank-2.5-lite 83.12%, GPT-5 ~71.71%, Gemini 2.5 Pro ~70.89%, Qwen3-32B ~69.54%. With a strong first stage, LLM rerankers *degrade*: baseline 81.58% โ†’ 80.63% (Qwen3-32B) / 79.49% (Gemini 2.0 Flash). Cost: "LLMs cost 25-60x more than rerank-2.5" ($1.25โ€“$3 vs $0.05 per 1M tokens); rerank-2.5 is "9x, 36x, and 48x faster than Claude Sonnet 4.5, GPT-5, and Gemini 2.5 Pro." Also: **listwise sliding-window beats single-pass long-context reranking by 26.6%, 25.27%, and 22.2%**. +**SOURCE:** Voyage AI โ€” https://blog.voyageai.com/2025/10/22/the-case-against-llms-as-rerankers/ โ€” 2025-10-22 +**STATUS:** vendor research + +**CLAIM:** Jina reranker v3 is **listwise, not pointwise and not late-interaction** โ€” a 0.6B Qwen3-based model doing "causal attention between the query and *all* candidate documents within a single context window," branded "last but not late" in explicit contrast to ColBERT. 131K context, up to 64 docs/pass. BEIR nDCG@10 **61.94**, "outperforming Qwen3-Reranker-4B while being 6ร— smaller." +**CLAIM:** v3.5 (0.6B) uses hybrid attention ("three sliding-window layers followed by two global layers") + self-distillation: BEIR **63.20** vs Qwen3-Reranker-4B 62.28; "+9.6 nDCG@10 over v3" on Struct-IR; long-context latency 16.1s โ†’ **10.3s**; prefill throughput 11.9k โ†’ 18.6k tokens/s (A100, FA-2, top-100 listwise, batch 1). +**SOURCES:** Jina AI โ€” https://jina.ai/news/jina-reranker-v3-0-6b-listwise-reranker-for-sota-multilingual-retrieval/ (2025-10-03) ยท https://jina.ai/news/jina-reranker-v3-5-faster-listwise-reranking-hybrid-attention-self-distillation (2026-08-03) +**STATUS:** GA + +**CLAIM:** Cohere split reranking into quality/latency SKUs: `rerank-v4.0-pro` for "state-of-the-art quality and complex use-cases" and `rerank-v4.0-fast` for "low latency and high throughput use-cases," 32k context, semi-structured (JSON) document support. +**SOURCES:** Cohere โ€” https://docs.cohere.com/changelog/rerank-v4.0 (undated) ยท https://cohere.com/blog/rerank-4 (2025-12-11, body would not render) ยท https://docs.cohere.com/docs/rerank +**STATUS:** GA โ€” **date conflict unresolved:** the Rerank 3.5 blog page reported 2024-12-02 on fetch while a search snippet claimed 2025-07-10. No v4.0 benchmark numbers were verifiable. + +**CLAIM:** Contextual AI ships an instruction-following reranker family at 1B/2B/6B (+ NVFP4 quantized), claiming "~35% increase in recency-awareness with our quantized 2B reranker compared to the second-best reranker" and beating all comparators on TREC 2025 Product Search "at a superior throughput, latency, and cost." Open weights on HuggingFace. Platform benchmark: reranker **61.2 on BEIR**, "outperforming the next best solution (Voyage-v2 at 58.3) by 2.9%"; full RAG agent "71.2% performance, a 5.4% improvement over the strongest baseline" (Cohere retrieval + Claude-3.5-Sonnet at 66.8%). +**SOURCES:** https://contextual.ai/blog/rerank-v2 (2025-08-27) ยท https://contextual.ai/blog/introducing-instruction-following-reranker (2025-03-11) ยท https://contextual.ai/blog/platform-benchmarks-2025 (2025-01-15) +**STATUS:** GA + open weights (vendor self-eval) + +**CLAIM:** MongoDB shipped a **native reranker inside the database** โ€” Voyage AI rerank-2.5, 32K context, "improves retrieval accuracy by up to 30%," public preview on MongoDB 8.3. Rationale is explicitly agentic: "In an agentic workflow, retrieval is iterativeโ€ฆ A weak result does not just hurt one response; it can send the next step off course and drive up cost." +**SOURCE:** MongoDB โ€” https://www.mongodb.com/company/blog/product-release-announcements/improving-agent-retrieval-native-reranking-hybrid-search โ€” 2026-07-02 +**STATUS:** preview + +**CLAIM:** Elastic's `text_similarity_reranker` retriever is the shipped mechanism (nested first-stage retriever + inference endpoint + `rank_window_size` + `min_score`). Elastic's stated value case is *calibrated scores enabling cutoffs*, not just ordering. +**SOURCE:** Elastic โ€” https://www.elastic.co/search-labs/blog/semantic-reranking-with-retrievers โ€” 2024-05-28 +**STATUS:** GA + +**CLAIM (cost warning):** Qdrant declines to endorse a default: "reranking can be slow. Processing millions of documents can take hours, which is why rerankers focus on refining results, not searching through the entire document collection." Names cross-encoder, ColBERT-multivector, and LLM rerankers as three valid families. +**SOURCE:** Qdrant โ€” https://qdrant.tech/documentation/search-precision/reranking-semantic-search/ โ€” undated +**STATUS:** documentation guidance + +### 3.5 Chunking, context compression, pruning + +**CLAIM:** Chroma's chunking eval found chunker choice moves recall by **up to 9%**, and smaller chunks win. ClusterSemanticChunker @200 tokens โ†’ recall 87.3%, precision 8.0%, IoU 8.0%; LLMChunker (GPT-4o) โ†’ recall 91.9% but precision/IoU 3.9%; RecursiveCharacterTextSplitter @200 no-overlap โ†’ recall 88.1%. Explicit callout: **OpenAI's documented default (800 tokens / 400 overlap) produced "slightly below-average recall and the lowest scores across all other metrics."** +**SOURCE:** Chroma โ€” "Evaluating Chunking Strategies for Retrieval" โ€” https://www.trychroma.com/research/evaluating-chunking โ€” 2024-07-03 +**STATUS:** vendor research report +*Direct contradiction: Elastic's `semantic_text` default is "250 words (approximately 400 tokens)"; OpenAI's is 800/400.* + +**CLAIM:** Jina's **late chunking** = encode the whole document with a long-context encoder first, then pool per chunk; no LLM in the loop. BeIR nDCG gains over naive chunking: NFCorpus 23.46 โ†’ 29.98, SciFact 64.20 โ†’ 66.10, TRECCOVID 63.36 โ†’ 64.70, FiQA2018 33.25 โ†’ 33.84. "The longer the document, the more effective the late chunking strategy becomes." +**SOURCE:** Jina AI โ€” https://jina.ai/news/late-chunking-in-long-context-embedding-models/ โ€” 2024-08-22 +**STATUS:** GA (in the jina-embeddings API) + +**CLAIM โ€” direct vendor-vs-vendor attack on Anthropic:** Jina explicitly frames Anthropic's contextual retrieval as inferior engineering โ€” "a brute-force approach" where "each chunk is sent to the LLM along with the full document" โ€” and claims late chunking has "No additional storage since the embedding size remains the same," is "Significantly faster than using an LLM to generate enrichment," and is "highly resilient to boundary cues" whereas Anthropic's method "relies on accurate and readable chunks." +**SOURCE:** Jina AI โ€” Han Xiao โ€” "What Late Chunking Really Is & What It's Not: Part II" โ€” https://jina.ai/news/what-late-chunking-really-is-and-what-its-not-part-ii/ โ€” 2024-10-03 +**STATUS:** vendor position piece +*No vendor published a head-to-head of late chunking vs contextual retrieval under matched conditions. No first-party vendor page was found shipping Anthropic-style contextual chunk augmentation as a GA feature.* + +**CLAIM โ€” the empirical basis everyone cites for pruning:** Chroma's context rot study across 18 models (Anthropic Opus 4 / Sonnet 4 / 3.7 / 3.5 / Haiku 3.5; OpenAI o3, GPT-4.1 family, GPT-4o, GPT-4 Turbo, GPT-3.5 Turbo; Google Gemini 2.5 Pro/Flash, 2.0 Flash; Alibaba Qwen3 235B-A22B / 32B / 8B) concludes models "do not use their context uniformly; instead, their performance grows increasingly unreliable as input length grows." Key retrieval findings: lower needle-question semantic similarity accelerates degradation; "**Even a single distractor reduces performance relative to the baseline**, and adding four distractors compounds this degradation further"; counterintuitively "models perform worse when the haystack preserves a logical flow of ideas"; on LongMemEval, "significantly higher performance on focused prompts compared to full prompts" (~300 tokens vs ~113k). Methodological attack: "long context evaluations for these models often demonstrate consistent performance across input lengths. However, these evaluations are narrow in scope and not representative of how long context is used in practice." +**SOURCE:** Chroma โ€” "Context Rot: How Increasing Input Tokens Impacts LLM Performance" โ€” https://www.trychroma.com/research/context-rot โ€” 2025-07-14 (Kelly Hong, Anton Troynikov, Jeff Huber) +**STATUS:** technical report โ€” *this is the single most-cited empirical justification for reranking, pruning, and top-k discipline in the 2025โ€“26 literature.* + +**CLAIM:** Chroma built context pruning *into the retrieval model itself*: **Context-1** is a 20B model (from gpt-oss-20B, SFT + RL) acting as a retrieval subagent that "can selectively discard tangential information" from its own context mid-search. Reported: web 0.88 final-answer-found, legal 0.89, email 0.92, finance 0.64 F1; "up to 10x faster" than frontier models; "400-500 tok/s end to end" on vLLM. Apache 2.0 weights. +**SOURCE:** Chroma โ€” https://www.trychroma.com/research/context-1 โ€” 2026-03-26 +**STATUS:** research-only, open weights + +**CLAIM:** Weaviate's position: pruning is mandatory โ€” "The worst memory system is the one that faithfully stores everything," requiring "periodic pruning, merging duplicates, deleting outdated facts." +**SOURCE:** Weaviate โ€” "Context Engineering โ€” LLM Memory and Retrieval for AI Agents" โ€” https://weaviate.io/blog/context-engineering โ€” 2025-12-09 +**STATUS:** vendor guidance + +**CLAIM:** LangChain reimplemented Anthropic's context-editing strategy as portable middleware. `SummarizationMiddleware` (`model`, `trigger` e.g. `("tokens", 4000)`, `keep` e.g. `("messages", 20)`) "persistently updates state by permanently replacing old messages with a summary." `ContextEditingMiddleware` with `ClearToolUsesEdit` "automatically prunes tool resultsโ€ฆ when the total input token count exceeds configured thresholds." Documented hazard: "When deleting messages, make sure that the resulting message history is valid" โ€” tool results must follow tool calls. *History: at LangChain 1.0 alpha (2025-09-08) middleware shipped with only HITL, summarization, and Anthropic prompt caching โ€” context editing was not in the initial set.* +**SOURCES:** https://docs.langchain.com/oss/python/langchain/short-term-memory ยท https://reference.langchain.com/python/langchain/agents/middleware/summarization/SummarizationMiddleware ยท https://reference.langchain.com/python/langchain/agents/middleware/context_editing/ContextEditingMiddleware ยท https://www.langchain.com/blog/agent-middleware (2025-09-08) +**STATUS:** stable in LangChain 1.x + +**CLAIM:** Claude Code compaction is **tiered**: "It clears older tool outputs first, then summarizes the conversation if needed. Your requests and key code snippets are preserved; detailed instructions from early in the conversation may be lost." Anti-thrash guard: "If a single file or tool output is so large that context refills immediately after each summary, Claude Code stops auto-compacting after a few attempts and shows an error instead of looping." Steerable via `/compact ` and a "Compact Instructions" section in CLAUDE.md. +**SOURCE:** Anthropic โ€” https://code.claude.com/docs/en/how-claude-code-works +**STATUS:** shipped-in-production + +**CLAIM:** Anthropic names context as the governing constraint of the whole product: "Most best practices are based on one constraint: Claude's context window fills up fast, and performance degrades as it fills." Named failure mode: "**The infinite exploration.** You ask Claude to 'investigate' something without scoping it. Claude reads hundreds of files, filling the context." Fix: scope narrowly or use subagents โ€” "Since context is your fundamental constraint, subagents are one of the most powerful tools available." +**SOURCE:** Anthropic โ€” https://code.claude.com/docs/en/best-practices +**STATUS:** shipped-in-production + +**CLAIM:** Anthropic says compaction alone is **insufficient** for long-horizon work โ€” the model can exhaust context mid-implementation and leave undocumented half-finished features. Recommended harness: initializer agent creating `init.sh`, `claude-progress.txt`, and an initial git commit; then coding agents that read those at session start and update them before ending; a structured JSON feature list with **"over 200 features"** marked pass/fail; "work on only one feature at a time" and self-verify end-to-end (they used Puppeteer MCP) before marking complete. Rule stated: "It is unacceptable to remove or edit tests because this could lead to missing or buggy functionality." +**SOURCE:** Anthropic โ€” "Effective harnesses for long-running agents" โ€” https://www.anthropic.com/engineering/effective-harnesses-for-long-running-agents โ€” 2025-11-26 (Justin Young et al.) +**STATUS:** guidance from production + +### 3.6 Grounding and citation verification + +**CLAIM:** Google's **check-grounding API** is a standalone, shipped verification step: returns "an overall support score of 0 to 1" approximating the fraction of claims grounded in supplied facts, plus claim-level citations mapped by byte position, optional per-claim scores (`enableClaimLevelScore`), and a boolean per claim for whether grounding is even required. Limits: answer โ‰ค4096 tokens, up to **200 facts** at โ‰ค10k chars each; `citation_threshold` defaults to **0.6**. Latency target **<500ms** so it can run inline. Grounding is strict โ€” a mostly-correct claim with one wrong date gets no citation. (Page notes Vertex AI Search is being renamed **Agent Search**.) +**SOURCE:** Google Cloud โ€” "Check grounding with RAG" โ€” https://docs.cloud.google.com/generative-ai-app-builder/docs/check-grounding โ€” accessed 2026-08-08 +**STATUS:** GA + +**CLAIM:** "Grounding with Google Search" supports **dynamic retrieval**: Gemini predicts whether a search would help, emitting a 0โ€“1 prediction score; `dynamicRetrievalConfig` threshold defaults to **0.7**, below which no search fires. The search-vs-no-search decision is a *scored gate*, not an agent judgment. +**SOURCE:** Google Cloud โ€” https://cloud.google.com/blog/products/ai-machine-learning/how-vertex-ai-grounding-helps-build-more-reliable-models โ€” `[snippet-only]` +**STATUS:** GA + +**CLAIM:** Anthropic runs citation verification as a **separate subagent** โ€” a `CitationAgent` processes documents and the drafted report post-research to identify specific locations for citations, "ensuring all claims are properly attributed to their sources." +**SOURCE:** Anthropic โ€” https://www.anthropic.com/engineering/multi-agent-research-system โ€” 2025-06-13 +**STATUS:** shipped-in-production + +**CLAIM:** Contextual AI ships **span-level** groundedness scoring, GA: "scores are reported for individual text spans allowing for precise detection of unsupported claims," returned in the API so developers can hide ungrounded claims or add caveats. Separately their Grounded Language Model (GLM) prioritizes retrieved knowledge over parametric knowledge via a `/generate` API. +**SOURCE:** Contextual AI โ€” https://contextual.ai/new/groundedness-scoring-of-model-responses-now-generally-available โ€” `[snippet-only; fetched page rendered nav only]` +**STATUS:** shipped API + +**CLAIM:** Azure AI Content Safety groundedness detection ships non-reasoning mode (fast binary for online use), reasoning mode (returns a `reasoning` field explaining ungrounded segments), and a correction feature returning `corrected Text` realigned to sources. Domains MEDICAL / GENERIC; tasks QnA / Summarization. +**SOURCE:** Microsoft โ€” https://learn.microsoft.com/en-us/azure/ai-services/content-safety/concepts/groundedness โ€” `[snippet-only]` +**STATUS:** shipped API (preview per quickstart) + +**CLAIM:** Vectara's HHEM is the longest-running public hallucination leaderboard: commercial **HHEM-2.3**, open **HHEM-2.1-Open**. Method: 7,700+ curated articles across news/tech/science/medicine/legal/sports/business/education, summarize using only document facts, temperature 0, refusals filtered. As of **2026-05-11**: Finix S1 32B 1.8%, GPT-5.4-nano 3.1%, Gemini 2.5 Flash Lite 3.3%. Measures **factual consistency**, explicitly not summary quality. +**SOURCE:** Vectara โ€” https://github.com/vectara/hallucination-leaderboard โ€” snapshot 2026-05-11 +**STATUS:** shipped API + open model + benchmark + +**CLAIM:** OpenAI enforces grounding at the **display layer** instead of the model layer: inline citations "must be made clearly visible and clickable," with a separate `sources` list of every URL consulted that "typically exceeds the number of inline citations." OpenAI ships **no faithfulness or groundedness grader** (see ยง5). +**SOURCE:** OpenAI โ€” https://developers.openai.com/api/docs/guides/tools-web-search +**STATUS:** GA + +### 3.7 Memory across turns + +| System | Model | Status | Notable | +|---|---|---|---| +| **Anthropic memory tool** | Client-side file CRUD under `/memories`; "just-in-time context retrieval" | **GA**, no beta header | Auto-injected system protocol includes "ASSUME INTERRUPTION." Pairs with compaction: "compaction keeps the active context smallโ€ฆ, memory preserves the information that must survive summarization." Path-traversal protection is the integrator's burden | +| **Claude Code auto memory** | "The first 200 lines or 25KB of MEMORY.md, whichever comes first, load at the start of each session" | shipped | Separate from CLAUDE.md; explicit warning: "Bloated CLAUDE.md files cause Claude to ignore your actual instructions!" | +| **LangMem / LangGraph Store** | semantic / episodic / procedural (procedural = evolved prompt instructions) | GA at release (2025-02-18) | Candid gap: "LangMem currently lacks opinionated utilities for this [episodic] type" | +| **Google ADK MemoryService** | Three backends: `InMemoryMemoryService` (keyword), `VertexAiMemoryBankService` (LLM extraction + consolidation), `VertexAiRagMemoryService` (vector similarity) | shipped | **The agentic-vs-fixed split appears again at the memory layer**: `load_memory` tool (agent-initiated) vs `preload_memory` tool (automatic at conversation start) | +| **Vertex Agent Engine Memory Bank** | "uses Generative AI models to generate memories" | **status not stated on the overview page โ€” do not assert GA** | ADK is the documented first-class path | +| **Mem0** | short-term (conversation history, working memory, attention context) + long-term (factual, episodic, semantic). "The search pipeline pulls from all layers, ranking user memories first, then session notes, then raw history" | shipped | LoCoMo **92.5** at mean **6,956 tokens** per retrieval vs "25,000+" full-context; Single Hop 94.6 / Multi-Hop 95.4 / Open-domain 82.3 / Temporal 92.5; "median latency stays flat at +1ms." Warning: "Avoid storing secrets or unredacted PIIโ€ฆ Mem0 is retrievable by design" | +| **Zep / Graphiti** | Bi-temporal knowledge graph; "Temporal edge invalidation instead of LLM summarization" for contradictions; vector + full-text + graph traversal "without requiring LLM-based reranking" | Graphiti OSS; Zep commercial (SOC 2, HIPAA, BYOC) | "Sub-200ms retrieval latency versus seconds to tens of seconds." LongMemEval: up to **18.5%** accuracy gain over full-context, "90% faster," "less than 2% of baseline tokens." DMR: Zep 94.8% vs MemGPT 93.4% โ€” with the honest caveat that GPT-4o full-context hit 98.2%, "suggesting DMR benchmark limitations" | +| **Letta** | V1 memory-blocks SDK **deprecating** in favor of Agent SDK with **MemFS** (git-tracked memory, "agent dreaming") | V1 deprecating | MemFS detail page 404'd โ€” specifics unverified | +| **CrewAI** | **Collapsed the taxonomy**: "CrewAI replaced separate memory types with a unified `Memory` class." LLM infers scope/categories/importance on save; retrieval ranks by semantic similarity + recency + importance | shipped | Default LanceDB at `./.crewai/memory`, OpenAI `text-embedding-3-large` (3072-d). Non-blocking saves, auto dedup, "deep recall" multi-step LLM analysis for complex queries; simple queries skip the LLM. *A direct counter-move to the semantic/episodic/procedural consensus.* | +| **Mastra** | Message History, Working Memory (injected as system message), Semantic Recall, and **Observational Memory** โ€” "background agents that compress old messages into dense observations" | shipped | Sub-agent delegation gets a fresh `threadId` and deterministic `resourceId` `{parentResourceId}-{agentName}` | +| **Pydantic AI** | **No memory abstraction, deliberately.** Just `message_history` between runs and typed `deps` via `RunContext` | shipped | "An agent run might represent an entire conversationโ€ฆ However, a conversation might also be composed of multiple runs" | + +**SOURCES:** https://platform.claude.com/docs/en/agents-and-tools/tool-use/memory-tool ยท https://code.claude.com/docs/en/how-claude-code-works ยท https://www.langchain.com/blog/langmem-sdk-launch (2025-02-18) ยท https://adk.dev/sessions/memory/ ยท https://docs.cloud.google.com/vertex-ai/generative-ai/docs/agent-engine/memory-bank/overview ยท https://docs.mem0.ai/core-concepts/memory-types and https://mem0.ai/research (2026-08-07) ยท https://blog.getzep.com/state-of-the-art-agent-memory/ (2025-01-22) and https://help.getzep.com/graphiti/getting-started/overview ยท https://docs.letta.com/concepts/letta ยท https://docs.crewai.com/en/concepts/memory ยท https://mastra.ai/docs/memory/overview ยท https://pydantic.dev/docs/ai/core-concepts/agent/ + +--- + +## SECTION 4 โ€” DEEP-RESEARCH / MULTI-HOP PRODUCTS + +### 4.1 Perplexity + +**CLAIM:** Own crawler and index over **200 billion unique URLs**, on "tens of thousands of CPUs and hundreds of terabytes of RAM," tens of thousands of indexing operations per second, ~200M daily queries. +**CLAIM โ€” pipeline shape:** explicitly multi-stage and progressive โ€” hybrid lexical+semantic retrieval โ†’ heuristic prefiltering for staleness โ†’ embedding-based scorers โ†’ **cross-encoder rerankers on the narrowed candidate set**. Scoring happens at both document and **sub-document span** level so agents get atomic units rather than whole pages. +**CLAIM โ€” latency:** median **358ms** / p95 **763ms**, vs competitors at 513โ€“1,375ms median and 808โ€“2,188ms p95. +**CLAIM โ€” quality (own open-sourced `search_evals`):** SimpleQA **.930**, FRAMES **.453** (single-step); BrowseComp **.371**, HLE **.288** (deep research mode). +**SOURCE:** Perplexity โ€” "Architecting and Evaluating an AI-First Search API" โ€” https://research.perplexity.ai/articles/architecting-and-evaluating-an-ai-first-search-api โ€” 2026-07-29 +**STATUS:** shipped-in-production + +**CLAIM:** Sonar Deep Research is "exhaustive searches across hundreds of sources." The docs' own sample run shows **21 search queries** and **193,947 reasoning tokens** for a single report, total **$0.816**. Perplexity prices the search *loop* separately from tokens: input $2/1M, output $8/1M, **citation tokens $2/1M**, **reasoning tokens $3/1M**, **search queries $5/1K**. In their sample, reasoning ($0.582) + searches ($0.105) dominated output ($0.091) by ~6x. +**SOURCE:** Perplexity โ€” Sonar Deep Research docs โ€” https://docs.perplexity.ai/docs/sonar/models/sonar-deep-research โ€” undated (accessed 2026-08-08) +**STATUS:** GA + +**CLAIM:** Perplexity's retrieval backend is **Vespa**. Vespa's CEO describes the priorities as completeness/freshness/speed, chunks as first-class retrieval units alongside documents, and "multiple stages of progressively advanced ranking" including cross-encoder rerankers. +**SOURCE:** Vespa โ€” Jon Bratseth โ€” "How Perplexity beat Google on AI Search with Vespa.ai" โ€” https://blog.vespa.ai/perplexity-show-what-great-rag-takes/ โ€” 2025-10-06 +**STATUS:** shipped-in-production + +**GAP FLAG:** Perplexity has published **no Comet orchestration architecture post**. Comet ships as two tiers (Assistant reads page context; Agent takes actions, with per-action permission gating), but no fan-out counts or loop description exist first-party. `perplexity.ai/hub/blog/*` returns 403 across the board. + +### 4.2 OpenAI Deep Research + +**CLAIM:** Deep research is "an early version of OpenAI o3 optimized for web browsing," trained with **end-to-end RL on browsing tasks** โ€” searching, clicking, scrolling, file interpretation, and sandboxed Python are *learned behaviors*, not a hand-written loop. Training mixed auto-gradable ground-truth tasks with open-ended rubric-graded tasks, scored by a chain-of-thought grader model. The model "pivots as needed in reaction to information it encounters" โ€” mid-trajectory replanning is trained. OpenAI explicitly warns citations may contain errors and that prompt injection encountered during browsing can alter model behavior. +**SOURCE:** OpenAI โ€” Deep Research System Card (Deployment Safety Hub) โ€” https://deploymentsafety.openai.com/deep-research โ€” 2025-02-25 +**STATUS:** shipped-in-production +*Note: `openai.com/index/introducing-deep-research/` returns 403; the commonly-cited HLE 26.6% / GAIA figures were NOT verified here and are not asserted.* + +**CLAIM:** The API exposes `o3-deep-research` and `o4-mini-deep-research`. At least one data source is mandatory: `web_search_preview`, `file_search` (**max 2 vector stores**), or a remote MCP server implementing `search`+`fetch`; `code_interpreter` optional. Output stream includes `web_search_call` actions (search, open_page, find_in_page), `code_interpreter_call`, `file_search_call`, `mcp_tool_call`, and a final `message` with inline citations (`annotations` carrying url, title, start_index, end_index). +**CLAIM โ€” termination is developer-controlled:** `max_tool_calls` explicitly "to constrain costs and latency." Background mode strongly recommended with webhooks (incompatible with ZDR). +**CLAIM โ€” no built-in clarification step.** OpenAI's own guidance is to preprocess the prompt with a cheaper model (docs reference `gpt-5.6`) before handing it to the deep research model. +**SOURCE:** OpenAI โ€” Deep research API guide โ€” https://developers.openai.com/api/docs/guides/deep-research โ€” accessed 2026-08-08 +**STATUS:** GA + +### 4.3 Anthropic Research + +**CLAIM:** Orchestrator-worker. Lead agent analyzes the query, develops strategy, spawns subagents exploring different aspects **in parallel** โ€” "3-5 subagents in parallel rather than serially," subagents themselves using "3+ tools in parallel." Parallelization "cut research time by up to 90% for complex queries." +**CLAIM โ€” hard-coded effort scaling in the orchestrator prompt:** simple fact-finding = **1 agent, 3-10 tool calls**; direct comparison = **2-4 subagents, 10-15 calls each**; complex research = **10+ subagents** with divided responsibilities. +**CLAIM โ€” search prompt principles:** breadth-first then narrow โ€” "start with short, broad queries, evaluate what's available, then progressively narrow focus." Extended thinking to plan; interleaved thinking after tool results to evaluate quality and identify gaps. Source-quality heuristics favor "specialized tools over generic ones" and primary sources over "SEO-optimized content farms" โ€” added after early agents "consistently chose" lower-quality sources. +**CLAIM โ€” numbers:** Opus 4 lead + Sonnet 4 subagents "outperformed single-agent Claude Opus 4 by **90.2%** on our internal research eval." Agents use "about 4ร— more tokens than chat interactions"; multi-agent "about **15ร—** more tokens than chats." "**Token usage by itself explains 80% of the variance**" in BrowseComp-style eval performance. Upgrading to Sonnet 4 gave larger gains than doubling the token budget on Sonnet 3.7. +**CLAIM โ€” stated limits:** not appropriate for tasks requiring all agents to share identical context, many agent-to-agent dependencies, or real-time coordination โ€” and "most coding tasks involve fewer truly parallelizable tasks than research." Known bottleneck: "Current lead-agent synchronous execution of subagents creates information-flow delays and prevents mid-research steering." Economics: "multi-agent systems require tasks where the valueโ€ฆ is high enough to pay for the increased performance." +**SOURCE:** Anthropic โ€” "How we built our multi-agent research system" โ€” https://www.anthropic.com/engineering/multi-agent-research-system โ€” 2025-06-13 (Hadfield, Zhang, Lien, Scholz, Fox, Ford) +**STATUS:** shipped-in-production + +**CLAIM (launch framing):** "Claude operates agentively, conducting multiple searches that build on each other while determining exactly what to investigate next," integrated with Google Workspace + web search, answers "in minutes," inline citations. +**SOURCE:** https://claude.com/blog/research โ€” 2025-04-15 +**STATUS:** beta at publication (Max/Team/Enterprise; US, Japan, Brazil); now shipped + +### 4.4 Microsoft + +**CLAIM:** M365 Copilot **Researcher** runs an explicit iterative loop โ€” Reasoning (pick next subtask + identify missing detail) โ†’ Retrieval (documents, emails, chats, calendar, transcripts, web) โ†’ Review (score relevance, write findings to a **scratch pad**) โ€” terminating on **diminishing returns**: it stops at iteration *m* when marginal insight ฮ”I_m < ฮต. Powered by OpenAI's deep research model combined with Copilot orchestration and "deep search." +**SOURCE:** Microsoft โ€” "Researcher agent in Microsoft 365 Copilot" โ€” https://techcommunity.microsoft.com/blog/microsoft365copilotblog/researcher-agent-in-microsoft-365-copilot/4397186 โ€” `[snippet-only; page renders title only. Treat the ฮ”I<ฮต formalism as attributed but not directly verified]` +**STATUS:** GA (Researcher and Analyst GA'd 2025-06-02) + +**CLAIM:** Microsoft frames Researcher's cost as intentional: "Researcher agent deliberately spends more time retrieving and analyzing," respects existing M365 permissions/policies, and may ask clarifying questions before researching. +**SOURCE:** Microsoft Learn โ€” https://learn.microsoft.com/en-us/microsoft-365/copilot/researcher-agent โ€” ms.date 2026-02-19, updated 2026-05-06 +**STATUS:** GA + +**CLAIM โ€” GraphRAG:** LLM-generated entity/relationship knowledge graph, bottom-up community detection with pre-generated community summaries, answering global questions ("top 5 themes") vector similarity cannot. Claimed to "consistently outperform baseline RAG" on comprehensiveness, source grounding, viewpoint diversity. **The launch post discusses no cost numbers at all.** +**SOURCE:** Microsoft Research โ€” https://www.microsoft.com/en-us/research/blog/graphrag-unlocking-llm-discovery-on-narrative-private-data/ โ€” 2024-02-13 +**STATUS:** research prototype โ†’ open source + +**CLAIM โ€” LazyGraphRAG (the walk-back):** defers LLM use โ€” NLP noun-phrase extraction instead of LLM entity extraction at index time, sentence-level relevance filtering before LLM processing, and "best-first and breadth-first search dynamics in an iterative deepening manner" at query time. Indexing cost "identical to vector RAG and **0.1% of the costs of full GraphRAG**"; "**more than 700 times lower query cost**" than GraphRAG Global Search at comparable quality; at **4%** of GraphRAG global-search spend it beats competing methods on both local and global queries. Eval: 5,590 AP news articles, 100 synthetic queries (50 local / 50 global). +**SOURCE:** Microsoft Research โ€” https://www.microsoft.com/en-us/research/blog/lazygraphrag-setting-a-new-standard-for-quality-and-cost/ โ€” 2024-11-25 +**STATUS:** research prototype (open sourced) + +### 4.5 Google + +**CLAIM:** Gemini Deep Research "creates a multi-step research plan for you to either revise or approve" (**human-in-the-loop planning**), then "continuously refines its analysis, browsing the web the way you do: searching, finding interesting pieces of information and then starting a new search based on what it's learned. It repeats this process multiple times." Uses the 1M-token context window; exports to Google Docs with source links. **Google does not publish a number of sites browsed.** +**SOURCE:** Google โ€” https://blog.google/products/gemini/google-gemini-deep-research/ โ€” 2024-12-11 +**STATUS:** shipped-in-production + +**CLAIM:** Google published a human-preference number rather than an orchestration number: raters preferred Gemini Deep Research reports over competitors "by more than a 2-to-1 margin." +**SOURCE:** Google โ€” https://blog.google/products/gemini/deep-research-gemini-2-5-pro-experimental/ โ€” 2025-04-08 +**STATUS:** shipped + +**CLAIM:** Built on Gemini 2.5 Pro, "optimized to perform task prioritization" and "able to identify when it reaches a dead-end when browsing" โ€” an explicit stopping/backtracking heuristic. HLE **7.95% (Dec 2024) โ†’ 26.9%, and 32.4% with higher compute (June 2025)**. +**SOURCE:** Google DeepMind / I/O 2025 โ€” https://blog.google/innovation-and-ai/models-and-research/google-deepmind/google-gemini-updates-io-2025/ โ€” 2025-05 โ€” `[snippet-only]` +**STATUS:** shipped + +**CLAIM:** Vertex AI RAG Engine is a six-stage managed pipeline (ingestion โ†’ transformation/chunking โ†’ embedding โ†’ indexing โ†’ retrieval โ†’ generation) with pluggable backends (RagManagedDb, Vector Search 2.0, Feature Store, Weaviate, Pinecone, Agent Platform Search) and a "Reranking for RAG" stage. **GA only in `europe-west3`/`europe-west4`; allowlist-GA in `us-central1`/`us-east1`/`us-east4`; preview in 14 other regions.** +**SOURCE:** Google Cloud โ€” https://docs.cloud.google.com/vertex-ai/generative-ai/docs/rag-engine/rag-overview โ€” accessed 2026-08-08 +**STATUS:** mixed GA/preview by region + +### 4.6 Glean and Exa + +**CLAIM:** Glean's Agentic Engine 2 has three published mechanisms: **adaptive planning** (intent interpretation โ†’ plan proposal โ†’ context grounding in the Enterprise Graph, with continuous re-planning), **tool orchestration** following an explicit "**explore โ†’ narrow โ†’ retrieve**" flow with sub-agent delegation, and **dual-layer memory** (session-level + persistent Personal/Enterprise Graph). +**Numbers:** **94% completeness** on end-to-end tasks; **19.4%** enumeration-quality gain over Engine 1; WritingBench **75.7% โ†’ 82.5%**; **20%** improvement on previously-downvoted queries; **21%** increase in Assistant query usage during testing; Enterprise Graph ingests "**3ร— more signals**." *Baseline is their own Engine 1.* +**SOURCE:** Glean โ€” https://www.glean.com/blog/live-fall-25-agentic-engine2-performance โ€” 2025-09-25 +**STATUS:** shipped-in-production + +**CLAIM:** Glean's earlier post describes a **Reflect** stage โ€” the agent self-assesses confidence in initial search results and decides whether additional tools are warranted *before* committing to the expensive agentic path, i.e. a cheap/expensive routing gate. Claimed "**24%** increase in relevance." Retrieval is hybrid (semantic + lexical + knowledge graph over 100+ connectors), role-based-permission scoped so unauthorized content never reaches the LLM. +**SOURCE:** Glean โ€” https://www.glean.com/blog/agentic-reasoning-future-ai โ€” 2024-11-19 +**STATUS:** shipped-in-production + +**CLAIM:** Exa Agent "divides the task into many subtasks and assigns subagents to research various domains at once," and uses "a fusion of frontier and cost-effective models to find the most cost-effective methodology" โ€” **model-tier routing inside the research loop**. Claims "up to **94%** reductions in token usage." WideSearch: ~50% Row-F1 at ~$0.50/query. +**CLAIM โ€” research budget as a first-class product dial:** fixed-price effort tiers `minimal` $0.012 / `low` $0.025 / `medium` $0.10 (default) / `high` $0.50 / `xhigh` $1.00 per request, plus metered `auto` (default cap $5) and `max` beta (cap $20). Billed as Agent Compute Units (1 ACU = $0.10) + $0.005 per search call. Async run model with polling/SSE/batch, replayable events, continuation from a completed run; outputs carry **field-level grounding** citations against a JSON `outputSchema`. +**SOURCES:** Exa โ€” https://exa.ai/blog/exa-agent (2026-06-16) ยท Agent API guide https://exa.ai/docs/reference/agent-api-guide (accessed 2026-08-08) +**STATUS:** GA (`max` effort in beta) + +**CLAIM:** Exa Deep Max published accuracy/latency frontier: Deep Search QA **90% at 64s** (vs You Frontier 84%/5908s, Parallel Ultra 8x 82%/1703s); FRAMES **94% in 11s** (vs Parallel Ultra 88%/1457s); HLE-Search **80% at 25s**, "matching GPT 5.4 on quality but at half the latency"; "up to 20x faster than the closest competitor." Architecture: "dozens of parallel calls to Exa Search," each "target[ing] a different angle," over an in-house index returning "results in under a second." Index described as "many petabytes," "semantic+lexical databases from scratch." +**SOURCE:** Exa โ€” https://exa.ai/blog/deep-max โ€” 2026-04-20 +**STATUS:** GA (pricing on request) + +### 4.7 Cross-cutting: orchestration shapes and stopping rules + +**The published designs split cleanly:** + +| Org | Shape | Stopping rule | +|---|---|---| +| **Anthropic Research** | Orchestrator + 3-5 (up to 10+) parallel subagents, each with its own context window, returning distilled 1-2k-token summaries | Model judgment steered by prompt heuristics; `max_uses` as a hard backstop | +| **OpenAI Deep Research** | One RL-trained agent, sequential trajectory | Developer-set hard cap (`max_tool_calls`) | +| **Azure AI Search agentic retrieval** | **No agent in the retrieval layer** โ€” one LLM planning pass, then stateless parallel subquery fan-out with per-subquery L2 reranking and a merge | **No stopping rule โ€” it is a single fixed fan-out round, not a loop** | +| **Microsoft Researcher** | Sequential reason/retrieve/review cycles with a scratch pad | Marginal-information threshold ฮ”I < ฮต | +| **Google Deep Research** | Human-approved plan, then iterative browse-and-refine | Trained dead-end detection + task prioritization | +| **Glean** | Adaptive planner + confidence-gated sub-agent delegation, explore โ†’ narrow โ†’ retrieve | Confidence gate before committing to the expensive path | +| **Exa Agent** | Subagents per domain with model-tier routing | Monetary cap (effort tier / ACU budget) | +| **Perplexity Sonar DR** | Query fan-out over own index, multi-stage rerank | Undisclosed; sample run = 21 queries | + +**Permissions as an architectural constraint (four orgs, independently):** Glean scopes retrieval to role-based ACLs so unauthorized content never reaches the model; OpenAI company knowledge "respects your existing company permissions"; Anthropic's argument for indexless code search includes avoiding permission-synchronization problems inherent to a centralized index; Elastic implements it as native document-level security in the query itself (zero cross-tenant leaks measured). + +**Citation verification, three different shapes:** Anthropic = a dedicated CitationAgent pass. Google = a standalone check-grounding API with a 0โ€“1 support score and per-claim "grounding required" flags at <500ms. OpenAI = display-layer enforcement plus a `sources` superset. Anthropic's web search makes citations non-optional and token-free; web fetch makes them opt-in and off by default. + +--- + +## SECTION 5 โ€” EVALS FOR AGENTIC RETRIEVAL + +### 5.1 The structural news: OpenAI is exiting the eval-product space + +**CLAIM:** Verbatim timeline: "June 3, 2026 | Deprecation announced for the Evals platform." / "Oct 31, 2026 | Existing evals become read-only." / "Nov 30, 2026 | The Evals dashboard and API are scheduled to shut down." Graders are also deprecated "as part of the evals and fine-tuning workflows they support." Recommended migration: OpenAI **Datasets** and โ€” remarkably โ€” third-party **Promptfoo**. +**SOURCES:** OpenAI โ€” Deprecations โ€” https://developers.openai.com/api/docs/deprecations (announced 2026-06-03) ยท Working with evals โ€” https://developers.openai.com/api/docs/guides/evals ยท Graders โ€” https://developers.openai.com/api/docs/guides/graders +**STATUS:** shipped feature being retired + +**CLAIM:** OpenAI's grader set was always thin: **String Check** (0/1), **Text Similarity** (BLEU/METEOR/ROUGE/fuzzy/cosine), **Score Model** (LLM returns a numeric score), **Python** (arbitrary code returning a float). **No faithfulness, groundedness, or citation grader ever existed.** There is no "label model" grader despite common belief. +**SOURCE:** same +**STATUS:** deprecating + +**CLAIM:** OpenAI's post-Evals agent story is **trace grading, not metrics**: "The fastest way to identify workflow-level issues" is trace grading; "Graders let you score those traces with structured criteria." No named agent metrics ship โ€” you write the grader. Stated dimensions (methodology only): tool selection, data/argument precision, agent handoff accuracy. Named anti-patterns: perplexity/BLEU, "vibe-based evals," deferring evals to production. +**SOURCES:** https://developers.openai.com/api/docs/guides/agent-evals ยท https://developers.openai.com/api/docs/guides/evaluation-best-practices +**STATUS:** shipped feature + methodology + +### 5.2 Is RAGAS still the default? + +**CLAIM:** Ragas repositioned away from "a bag of reference-free RAG metrics" toward an experiments-first loop โ€” the docs home headlines moving "from 'vibe checks' to systematic evaluation loops," foregrounding `experiments`, custom metrics via decorators, and dataset/result tracking. RAG is now **one of seven metric families** (RAG; NVIDIA metrics; Agents/Tool Use; Natural Language Comparison; SQL; General Purpose incl. Aspect Critic and Rubrics; Summarization). +**SOURCES:** https://docs.ragas.io/en/stable/ (2025-12-09) ยท https://docs.ragas.io/en/stable/concepts/metrics/available_metrics/ (2025-12-09) ยท https://docs.ragas.io/en/stable/concepts/experimentation/ (2025-12-09) +**STATUS:** open-source library +*Notably, the experiments docs give **no** guidance on when prefab metrics stop being sufficient โ€” the crux of the disagreement below.* + +**CLAIM:** Ragas' agentic metrics: Topic Adherence (precision/recall/F1 vs `reference_topics`), Tool Call Accuracy (0โ€“1; strict-order default or flexible), Tool Call F1 (order-independent), Agent Goal Accuracy (**binary 0/1**, with or without reference). +**SOURCE:** https://docs.ragas.io/en/stable/concepts/metrics/available_metrics/agents/ โ€” 2025-12-09 +**STATUS:** open-source + +**CLAIM โ€” the flat "No."** Asked "Should I use 'ready-to-use' evaluation metrics?" Hamel Husain and Shreya Shankar answer **"No."** Supporting: "Generic evaluations waste time and create false confidence"; "These metrics measure abstract qualities that may not matter for your use case"; "All you get from using these prefab evals is you don't know what they actually do." +**IMPORTANT ACCURACY NOTE:** RAGAS is **not named** in the FAQ. The critique is categorical ("prefab evals"), not vendor-specific. The only vendors named are observability platforms: "Vendors I encounter the most organically in my work are: Langsmith, Arize and Braintrust." Do not attribute a named anti-Ragas quote to Hamel. +**SOURCE:** Hamel Husain & Shreya Shankar โ€” AI Evals FAQ โ€” https://hamel.dev/blog/posts/evals-faq/ โ€” published 2025-05-28, last modified **2026-07-18** +**STATUS:** methodology recommendation + +**CLAIM โ€” the clearest replacement paradigm:** Contextual AI's **LMUnit** = natural-language unit tests. A unit test is "A specific, clear, testable statement or question in natural language about a desirable quality of an LLM's response." Continuous 1โ€“5 score, `POST https://api.contextual.ai/v1/lmunit`, inputs `query`/`response`/`unit_test`, 7,000-token cap. Claims: SOTA on FLASK and BigGenBench; top-5 on RewardBench at 93.5%; beat GPT-4o and Claude 3.5 Sonnet "by over 9%" on in-house finance/engineering data. +**SOURCES:** https://contextual.ai/lmunit/ (2024-12-18; open-sourced 2025-07-22) ยท https://docs.contextual.ai/api-reference/lmunit/lmunit +**STATUS:** shipped API + open model + +### 5.3 The practitioner canon (Hamel, Jason Liu, Eugene Yan) + +**CLAIM (Hamel, 2024):** "Don't rely on generic evaluation frameworks to measure the quality of your AI. Instead, create an evaluation system specific to your problem." Three levels: unit tests/assertions โ†’ human & model eval โ†’ A/B testing. And: "Many vendors want to sell you tools that claim to eliminate the need for a human to look at the data." +**SOURCE:** https://hamel.dev/blog/posts/evals/ โ€” 2024-03-29 + +**CLAIM (Hamel, 2024 โ€” Critique Shadowing):** "If your evaluations consist of a bunch of metrics that LLMs score on a 1-5 scale (or any other scale), you're doing it wrong." Method: a single **principal domain expert** issues binary pass/fail + written critique; iterate the judge to alignment (Honeycomb case: ">90% agreement between the LLM and Phillip" in three iterations). On generic judges: "Nothing is strictly wrong with them. It's just that many people are led astray by them." Track precision and recall separately, not raw agreement. +**SOURCE:** https://hamel.dev/blog/posts/llm-judge/ โ€” 2024-10-29 + +**CLAIM (Hamel, 2025):** Error analysis is "the single most valuable activity in AI development and consistently the highest-ROI activity." Generic metrics "create a false sense of measurement and progress" and "fragment your attention." Custom annotation tooling: "every domain has unique needs that off-the-shelf tools rarely address"; teams with good data viewers "iterate 10x faster." +**SOURCE:** https://hamel.dev/blog/posts/field-guide/ โ€” 2025-03-24 + +**CLAIM (Hamel & Shreya โ€” the most citable operational numbers in the field):** Error-analysis pipeline: representative dataset โ†’ open coding with domain experts โ†’ axial coding into a failure taxonomy โ†’ iterate to theoretical saturation, "**~100 traces minimum**"; "if ~20 traces don't turn up a new category, you can stop"; weekly "review 10-20 tracesโ€ฆ focusing on outliers"; "at least 100+ fresh traces each review cycle." Binary over Likert because Likert lets people "hide uncertainty in middle values" and needs larger samples. Judge validation: "Focus on achieving high True Positive Rate (TPR) and True Negative Rate (TNR) with your judge on a held out labeled test set," then "correct its estimates to determine the actual failure rate." Judge model: "Using the same model is usually fine because the judge is doing a different task than your main LLM pipeline." Custom annotation UI is "the single most impactful investment you can make." +**CLAIM (retrieval-specific, and they DO prescribe classical IR metrics):** Recall@k, Precision@k, MRR; build eval sets by "reverse-generating queries from documents." Synthetic query generation must use **structured dimensions** โ€” define variation categories โ†’ hand-write ~20 tuples โ†’ two-step generation tupleโ†’NL query โ€” explicitly *not* unstructured "give me test queries." Caveats: synthetic data fails for domain-specific content, low-resource languages, high-stakes domains. +**CLAIM (agent eval):** two-phase โ€” end-to-end task success as a black box, then step-level diagnostics using a **transition failure matrix** mapping last successful state vs first failure location to find workflow hotspots. +**SOURCE:** https://hamel.dev/blog/posts/evals-faq/ โ€” pub 2025-05-28, mod 2026-07-18 + +**CLAIM (Jason Liu, 2024 โ€” the RAG flywheel):** 9 stages: Initial Implementation โ†’ Synthetic Data Generation โ†’ Fast Evaluations โ†’ Real-World Data Collection โ†’ Classification/Analysis โ†’ System Improvements โ†’ Production Monitoring โ†’ User Feedback โ†’ Iteration. "Generate synthetic questions for each chunk of text in your database. Use these to test your retrieval system and calculate precision and recall scores." Retrieval evals are "lightning-fast (milliseconds vs. seconds per question)." Leading indicators: retrieval experiments run per week, precision/recall improvement on synthetic data, time to run the eval suite. Segmentation: unsupervised clustering of real questions into topics, few-shot topic classifiers, plus an **"Other" bucket monitored over time** to detect drift in user needs. +**SOURCE:** https://jxnl.co/writing/2024/08/19/rag-flywheel/ โ€” 2024-08-19 + +**CLAIM (Jason Liu, 2025):** "If your retrieval is wrongโ€”pulling the wrong chunk entirelyโ€”no model version will fix that." Distinguishes **inventory gaps** (data missing) from **capability gaps** (data present, not surfaced). Feedback UX: thumbs up/down plus "Highlight which snippet is wrong or missing," feeding embedding/reranker training. +**SOURCE:** https://jxnl.co/writing/2025/01/24/systematically-improving-rag-applications/ โ€” 2025-01-24 + +**CLAIM (Jason Liu, 2025 โ€” "There Are Only 6 RAG Evals"):** Tier 2: Context Relevance (C|Q, retrieval), Faithfulness/Groundedness (A|C, generation), Answer Relevance (A|Q, end-to-end). Tier 3: Context Support Coverage (C|A), Question Answerability (Q|C), Self-Containment (Q|A). Beneath these: "Retrieval Precision & Recallโ€ฆ don't require LLMs, and provide quick feedback for retriever tuning" โ€” MAP@K and MRR@K. Critique: "Don't waste time on complexity theater"; "teams obsess over generation quality while neglecting to ensure retrieval (C|Q) is even working correctly." +**SOURCE:** https://jxnl.co/writing/2025/05/19/there-are-only-6-rag-evals/ โ€” 2025-05-19 + +**CLAIM (Eugene Yan):** "I tend to be skeptical of correlation metricsโ€ฆ where possible, I have my evaluators return binary outputs." Objective tasks (factuality, toxicity) โ†’ direct scoring; subjective tasks (tone, persuasiveness) โ†’ pairwise. Documented judge biases: position bias (gpt-3.5 ~50%, claude-v1 ~70%), verbosity bias (both preferred longer responses ">90% of the time"), self-enhancement (gpt-4 +10% own-output win rate; claude-v1 +25%). Faithfulness correlation for gpt-4 only ฯโ‰ˆ0.55; **HaluEval best model 58.5% accuracy**. +**SOURCE:** https://eugeneyan.com/writing/llm-evaluators/ โ€” 2024-08 + +**CLAIM (Eugene Yan โ€” AlignEval):** "Align AI to human. Calibrate human to AI. Repeat." Label โ‰ฅ20 samples pass/fail *before* writing criteria โ€” "working backward from the data" so criteria reflect real, frequent defects. Reports sample size, recall, precision, F1, Cohen's ฮบ, TP/FP/TN/FN. +**SOURCE:** https://eugeneyan.com/writing/aligneval/ โ€” 2024-10 + +**CLAIM (Jo Kristian Bergum โ€” the practical middle path):** calibrate an LLM judge against a small human-labeled set, then scale it into a Cranfield-style test collection. 90 human-labeled query-passage pairs over 26 queries โ†’ calibrate GPT-4o โ†’ then label **10,372 unique query-passage pairs from 386 real queries**. "There are few cases where the LLM disagrees by more than one level. In only 1 case does it assign irrelevant for something that the human assigned as highly relevant." +**SOURCE:** Vespa โ€” https://blog.vespa.ai/improving-retrieval-with-llm-as-a-judge/ โ€” 2024-07-03 + +### 5.4 Anthropic's published eval position (the most conservative shipped stance) + +**CLAIM:** An eval is "a test for an AI system: give an AI an input, then apply grading logic to its output to measure success." Three grader classes with explicit tradeoffs โ€” code (fast/cheap/objective/reproducible but "brittle to valid variations"), model-based (flexible/scalable, "non-deterministic," more expensive), human ("gold standard quality" but expensive/slow). Why agents are harder: "Agents use tools across many turns, modifying state in the environment and adapting as they goโ€”which means mistakes can propagate and compound." +**CLAIM โ€” sample size:** "**20-50 simple tasks drawn from real failures is a great start.**" +**CLAIM โ€” judge calibration:** "Model-based graders should be closely calibrated with human experts to gain confidence that there is little divergence"; "give the LLM a way out, like providing an instruction to return 'Unknown'." +**CLAIM โ€” the killer line:** "**As a rule, we do not take eval scores at face value until someone digs into the details of the eval and reads some transcripts.**" +**CLAIM โ€” search-specific:** they built "evals covering both directions: queries where the model should searchโ€ฆ and queries where it should answer from existing knowledge." +**CLAIM โ€” lifecycle:** automated evals pre-launch/CI-CD โ†’ production monitoring for distribution drift โ†’ A/B testing once traffic suffices โ†’ continuous user-feedback triage + transcript review. +**SOURCE:** Anthropic โ€” "Demystifying evals for AI agents" โ€” https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents โ€” 2026-01-09 +**STATUS:** methodology recommendation (first-party) + +**CLAIM (the research-agent rubric, verbatim criteria):** "factual accuracy (do claims match sources?), citation accuracy (do the cited sources match the claims?), completeness (are all requested aspects covered?), source quality (did it use primary sources over lower-quality secondary sources?), and tool efficiency (did it use the right tools a reasonable number of times?)." Single LLM call, 0.0โ€“1.0 plus pass/fail. Started with **~20 queries**. Prioritized "**end-state evaluation rather than turn-by-turn analysis**" because "agents could follow different valid paths." Human testing caught what automation missed โ€” notably an **SEO-optimized-content bias over authoritative academic sources**, a directly agentic-retrieval failure mode. +**SOURCE:** https://www.anthropic.com/engineering/multi-agent-research-system โ€” 2025-06-13 + +### 5.5 Agent-specific eval as a shipped product + +**CLAIM โ€” Microsoft Foundry ships the most granular suite, split System vs Process, and it is deliberately BINARY.** Verbatim: agent evaluators "function like unit tests for agentic systemsโ€”they take agent messages as input and output binary Pass/Fail scores (or scaled scores converted to binary scores based on thresholds)." + +| Evaluator | Class | Output | +|---|---|---| +| Task Completion (preview) | System | Binary | +| Customer Satisfaction (preview) | System | **1โ€“5 Likert** (the lone exception) | +| Task Adherence (preview) | System | Binary | +| Task Navigation Efficiency | System | Binary + precision/recall/F1 (`exact_match` / `in_order_match` / `any_order_match`) | +| Intent Resolution (preview) | System | Binary via threshold on 1โ€“5 | +| Tool Call Accuracy | Process | Binary via threshold on 1โ€“5 | +| Tool Selection | Process | Binary | +| Tool Input Accuracy | Process | Binary (6 strict criteria) | +| Tool Output Utilization | Process | Binary | +| Tool Call Success | Process | Binary | +| Quality Grader (preview) | Quality | Binary | + +Recommended judge model: `gpt-5-mini`. +**DOCUMENTED BLIND SPOT:** Foundry's own docs say to **avoid** `tool_call_accuracy`, `tool_input_accuracy`, `tool_output_utilization`, `tool_call_success`, and `groundedness` when the agent calls Azure AI Search, Bing Grounding, Bing Custom Search, SharePoint Grounding, Code Interpreter, Fabric Data Agent, or Web Search โ€” i.e. **exactly the agentic-retrieval tools are the least-supported case.** +**SOURCE:** Microsoft โ€” https://learn.microsoft.com/en-us/azure/foundry/concepts/evaluation-evaluators/agent-evaluators โ€” ms.date 2026-06-02 +**STATUS:** shipped (several in public preview) + +**CLAIM โ€” LangSmith/AgentEvals ships four deterministic trajectory-match modes plus an LLM-judge trajectory evaluator.** `create_trajectory_match_evaluator` modes: **strict** (exact messages+tool calls, same order), **unordered** (same tool calls, any order), **subset** (agent calls only tools from reference, no extras), **superset** (reference tools plus extras allowed). LLM-judge variant "can assess nuanced aspects like efficiency and appropriateness," reference trajectory optional. +**SOURCE:** https://docs.langchain.com/langsmith/trajectory-evals +**STATUS:** OSS + shipped + +**CLAIM:** LangSmith shipped **Insights Agent** (clusters production traces to surface failure modes โ€” "Group by poor interactions: Cluster based on how your agent is messing up") and **Multi-turn Evals** scoring whole conversations on "Semantic intentโ€ฆ Semantic outcomesโ€ฆ Agent trajectory," firing online when a conversation concludes. *This is vendor productization of Hamel's error analysis.* +**SOURCE:** https://www.langchain.com/blog/insights-agent-multiturn-evals-langsmith โ€” 2025-10-23 +**STATUS:** shipped + +**CLAIM โ€” Databricks MLflow 3 ships single-turn AND multi-turn judges.** Single-turn: relevance, retrieval quality, safety, groundedness, correctness, `ToolCallCorrectness`, `ToolCallEfficiency` (trace-based). Retrieval judges are separate first-class scorers: `RetrievalRelevance`, `RetrievalGroundedness`, `RetrievalSufficiency` (requires ground truth). **Multi-turn: conversation completeness, user frustration, knowledge retention across interactions, role adherence, safety across the conversation.** +**SOURCES:** https://docs.databricks.com/aws/en/mlflow3/genai/eval-monitor/concepts/judges/ (2026-06-23) ยท https://mlflow.org/docs/latest/genai/eval-monitor/scorers/llm-judge/predefined/ ยท https://docs.databricks.com/aws/en/generative-ai/agent-evaluation/ (2026-07-28, recommends migrating to MLflow 3) +**STATUS:** shipped + +**CLAIM โ€” the strongest vendor-side concession that generic judges are inadequate:** MLflow ships judgeโ†’human **alignment** as a product with three optimizers โ€” **MemAlign** (default, "dual-memory system for fast and cheap few-shot alignment"), **SIMBA** (DSPy), **GEPA** (DSPy reflection). Requires "minimum 10" human-labeled traces, recommends โ‰ฅ30% each positive and negative. Claim: "**Aligned judges show 30-50% reduction in false positives/negatives compared to generic evaluation prompts.**" +**SOURCE:** https://mlflow.org/docs/latest/genai/eval-monitor/scorers/llm-judge/alignment/ +**STATUS:** shipped + +**CLAIM:** LlamaIndex ships retrieval eval **separately** from response eval โ€” retrieval: MRR and hit-rate with synthetic question-context pair generation from unstructured text; response: Faithfulness, Answer/Context Relevancy, Correctness, Semantic Similarity, Guideline adherence. Explicitly notes "many of these current evaluation modules do *not* require ground-truth labels." +**SOURCE:** https://developers.llamaindex.ai/python/framework/module_guides/evaluating/ +**STATUS:** OSS + +### 5.6 Retrieval-stage metrics and the label problem + +**CLAIM:** Consensus metric set is Precision@K, Recall@K, MRR@K, MAP@K, NDCG@K, with rank-awareness as the selection criterion. "NDCG@K is highly popular for evaluating retrieval systems" and is the default for the MTEB Retrieval category. +**SOURCE:** Weaviate โ€” https://weaviate.io/blog/retrieval-evaluation-metrics โ€” 2024-05-28 + +**CLAIM:** "It beginsโ€ฆ with offline metrics to predict the system's performance before deployment." Cites Spotify using Recall@K during training then MRR@K on larger eval sets. +**SOURCE:** Pinecone โ€” https://www.pinecone.io/learn/offline-evaluation/ โ€” 2023-06-30 + +**CLAIM โ€” the label-coverage problem, quantified:** "A single summary statistic misses many nuances to evaluating search." Elastic computes a **judge rate** โ€” "the average percentage of the top-10 documents that have a score in the qrels file." Using Phi-3-mini-4k to fill unjudged documents they found **57.6% of tested unlabeled docs were actually relevant**, with LLMโ€“human agreement "close to 80%." BEIR is described as the community's nDCG@10 "holy grail." +**SOURCE:** Elastic Search Labs โ€” https://www.elastic.co/search-labs/blog/evaluating-search-relevance-part-1 โ€” 2024-07-16 + +**CLAIM โ€” public benchmarks don't predict production:** MTEB/BEIR are "generic, which fail to capture the domain-specificity of real-world retrieval applications," "overly clean," and contaminated โ€” "Models have already seen most of these benchmarks in their training data." 11.91% of Wikipedia English query pairs exceeded 0.9 cosine similarity to LLM reproductions. **Ranking inversion demonstrated:** jina-embeddings-v3 beats text-embedding-3-large "across all MTEB English tasks," but on the WandBot production corpus Recall@10 was voyage-3-large 0.679 > text-embedding-3-large 0.602 > jina-embeddings-v3 0.532. LLM-judge/human alignment reached 75.2% after iteration. +**SOURCE:** Chroma โ€” "Generative Benchmarking" โ€” https://www.trychroma.com/research/generative-benchmarking โ€” 2025-04-07 (Hong, Troynikov, Huber, with Morgan McGuire of W&B) +**STATUS:** technical report โ€” *this quietly invalidates the BEIR/nDCG@10 leaderboard arms race that Cohere, Voyage, Jina, Elastic, and Contextual AI all use as their primary marketing claim.* + +### 5.7 Benchmarks vendors actually cite in 2025โ€“26 + +| Benchmark | What it measures | Detail | Maintainer | +|---|---|---|---| +| **BRIGHT** | Reasoning-intensive retrieval โ€” the benchmark that broke the "embeddings are solved" narrative | 1,385 queries from StackExchange, LeetCode, math competitions, requiring "in-depth reasoning to identify relevant documents," vs prior benchmarks where "keyword or semantic-based retrieval is usually sufficient." Metric nDCG@10. Leaderboard: Mira-Reasoning-Retrieval 66.9 (2026-04-22), INF-X-Retriever 63.4 (2025-12-20), RakanEmbed4B 52.4 (2026-03-20) | HKU, Princeton, UW, **Google Cloud AI Research** โ€” https://brightbenchmark.github.io/ | +| **BEIR** | Zero-shot IR baseline; nDCG@10 lingua franca | 18 public datasets; NDCG/MAP/Recall/Precision at k โˆˆ {1,3,5,10,100,1000} | Thakur, Reimers, Gurevych, Lin, Rรผcklรฉ โ€” https://github.com/beir-cellar/beir | +| **BrowseComp** | Browsing-agent persistence and creativity | **1,266 questions**, short verifiable answers. Wei, Sun, Papay, McKinney, Han, Fulford, Chung, Passos, Fedus, Glaese | OpenAI โ€” arXiv:2504.12516 (2025-04-16); openai.com/index/browsecomp/ 403s | +| **FRAMES** | Unified factuality + retrieval + reasoning | 824 multi-hop questions, 2โ€“15 Wikipedia articles each. **Baselines: Gemini-Pro-1.5 40.8% naive, 47.4% BM25@4, 66% multi-step retrieval+reasoning, 72.9% oracle retrieval** โ€” a ~26-point retrieval gap, the strongest single argument for evaluating retrieval separately | Google โ€” arXiv:2409.12941 (2024-09) `[snippet-only]` | +| **TREC RAG** | Academic gold standard | 2026 tasks: Retrieval (R) and RAG. **Corpus changed for 2026 to ClimbMix-400b (NVIDIA), replacing MS MARCO v2.1**, via Pyserini REST. "RAG25 nuggets" as dev data; **ResearchRubrics** for system testing; **RAGDoll** as the automated end-to-end eval framework. 2026: topics July 6, submissions Aug 8, conference Nov | https://trec-rag.github.io/ | +| **DeepResearch Bench** | Deep-research agents | 100 expert-crafted tasks (50 EN / 50 ZH), 22 fields, derived from 96,000+ user queries. **RACE** (comprehensiveness, insight/depth, instruction adherence, clarity) + **FACT** (extracts factual claims, verifies cited sources, computes citation accuracy). **As of May 2026 migrated to GPT-5.5 as primary evaluator, replacing Gemini-2.5-Pro** | USTC โ€” https://github.com/Ayanami0730/deep_research_bench | +| **MMTEB / MTEB v2** | 500+ tasks, 250+ languages | Adds instruction following, long-document retrieval, code retrieval. Headline: smaller multilingual models often outperform large LLMs | ICLR 2025, arXiv:2502.13595 `[snippet-only; leaderboard Space returned loading shell]` | +| **Vectara HHEM** | Factual consistency (not summary quality) | See ยง3.6 | https://github.com/vectara/hallucination-leaderboard | + +### 5.8 Online / production eval + +**CLAIM:** Microsoft Foundry ships four mechanisms: "Continuous evaluation: Quality and safety evaluation of production traffic **at a sampled rate**"; "Scheduled evaluationโ€ฆ using test datasets **to detect system drift**"; "Scheduled red teaming"; "Azure Monitor alerts." Plus **cluster analysis** to group evaluation failures. OpenTelemetry tracing supporting LangChain, LangGraph, OpenAI Agents SDK, Microsoft Agent Framework. Playground evals are **on by default and consumption-billed**. +**SOURCE:** https://learn.microsoft.com/en-us/azure/foundry/concepts/observability โ€” ms.date 2026-07-31, updated 2026-08-01 + +**CLAIM:** LangSmith online evals: configurable sampling (0.1 = 10% of filtered runs); weekly LLM spend limits per project/dataset that auto-pause evaluators; filters to trigger **only on runs with negative user feedback**, specific tool invocations, or metadata (e.g. customer tier); backfill rules onto historical runs. +**SOURCE:** https://docs.langchain.com/langsmith/online-evaluations + +**CLAIM:** Braintrust โ€” "Online scoring evaluates production traces automatically as they're logged, running asynchronously with no impact on latency." Model = Data + Task + Scorers/Classifiers, where "scorers measure quality with numeric scores, while classifiers apply categorical labels." +**SOURCE:** https://www.braintrust.dev/docs/guides/evals + +**CLAIM:** Arize Phoenix uses "function calling (tool use) to extract structured judgments" so the judge emits a **categorical label** which Phoenix "mapsโ€ฆ to its numeric score" โ€” label-first, score-derived, matching the Hamel/Yan binary preference. Native OpenTelemetry, "up to 20x performance gains via concurrency and batching," explanations on all LLM evals. +**SOURCE:** https://arize.com/docs/phoenix/evaluation/llm-evals + +**CLAIM:** Galileo's alignment mechanisms are **CLHF** ("Use CLHF to continuously improve the metrics") and **Autotune** ("continuously provide feedback in natural language that automatically improves the metrics"), with **Luna-2** small models as the metric-computation engine. RAG metrics: chunk relevance, context relevance, context precision, Precision@K, chunk attribution, chunk utilization, context adherence, completeness. +**SOURCE:** https://docs.galileo.ai/concepts/metrics/overview + +--- + +## SECTION 6 โ€” WHERE VENDORS DISAGREE (consolidated) + +**D1 โ€” Code retrieval: embeddings vs grep vs deterministic index. The sharpest live fight.** + +| Party | Position | Evidence | Date | +|---|---|---|---| +| **Cursor** | Embeddings are *necessary* at scale; use both. "semantic search is currently necessary to achieve the best results, especially in large codebases" / "the combination of these two leads to the best outcomes" | Offline +12.5% avg QA accuracy (range 6.5%โ€“23.5%); online A/B +0.3% code retention overall, **+2.6% on codebases with 1,000+ files**, โˆ’2.2% dissatisfied follow-ups. Trained a custom embedding model on agent session traces | https://cursor.com/blog/semsearch โ€” 2025-11-06 | +| **Cognition** | *Both* plain embeddings and plain agentic search are wrong. Embedding search gives "inaccurate results for complex queries" and "context pollution"; agentic search needs "dozens of sequential roundtrips." Fix: SWE-grep, an RL-trained retrieval model doing up to 8 parallel tool calls over max 4 turns | SWE-grep-mini 2,800+ tok/s (20x Haiku 4.5's 140); SWE-grep 650+ tok/s. "An order of magnitude faster" while "matching or outperforming" frontier models. 5-second "flow window" design target | https://cognition.com/blog/swe-grep โ€” 2025-10-16; https://cognition.com/blog/swe-1-5 โ€” 2025-10-29 | +| **Sourcegraph** | **Two positions 7 months apart.** Oct 2025: "Semantic search complements keyword search. It doesn't replace itโ€ฆ the best results come from using both." May 2026: embedding retrieval "returns plausible-looking results that miss cross-cutting impact"; on large codebases "agents ship plausible-looking code with latent bugs." Argues for "exact symbol definitions, callsites, and implementers" | Argument, not benchmark. Ships MCP Server, Deep Search, Code Search ("Deterministic, exact, exhaustive engine") | https://sourcegraph.com/blog/semantic-code-search-what-it-is-and-how-it-works โ€” 2025-10-06; https://sourcegraph.com/blog/agentic-coding โ€” 2026-05-21 | +| **Anthropic** | Product stance: "RAG-powered AI coding toolsโ€ฆ embed the entire codebaseโ€ฆ At large scale, those systems can fail because embedding pipelines can't keep up with active engineering teams." Claude Code traverses the filesystem, greps, follows references โ€” "no embedding pipeline or centralized index to maintain." But the nuanced version concedes: "Semantic search is usually faster than agentic search, but less accurateโ€ฆ start with agentic search, and only add semantic search if you need faster results" | No published retrieval benchmark | https://claude.com/blog/how-claude-code-works-in-large-codebases-best-practices-and-where-to-start โ€” 2026-05-14; context engineering post 2025-09-29 | +| **LlamaIndex** | "agents with good filesystem toolsโ€ฆ outperform naive semantic search" | Assertion | 2026-03-03 | + +**Direct contradiction:** Cursor says semantic search is *necessary* specifically on large codebases; Sourcegraph says approximate retrieval fails specifically on large codebases. Both ship products; both have commercial interest (Cursor trained its own embedding model, Sourcegraph sells a deterministic index). +**ACCURACY NOTE:** No first-party Anthropic page says "we removed vector search from Claude Code" or "embeddings don't work for code." That claim circulates in secondary sources. Anthropic's architecture reveals the position; the prose is more hedged. And Anthropic's own pro-embeddings artifact (Contextual Retrieval) has never been retracted โ€” the honest reconciliation is **by domain**: embeddings for document corpora, agentic search for code. + +**D2 โ€” Multi-agent vs single agent. Published one day apart.** +- **Cognition (2025-06-12):** "running multiple agents in collaboration only results in fragile systems." Two principles: "Share context, and share full agent traces, not just individual messages" and "Actions carry implicit decisions, and conflicting decisions carry bad results." Names OpenAI Swarm and Microsoft AutoGen as libraries that "actively push concepts which I believe to be the wrong way." Calls context engineering "the #1 job of engineers building AI agents." โ€” https://cognition.com/blog/dont-build-multi-agents +- **Anthropic (2025-06-13):** multi-agent beat single-agent by **90.2%** on their internal research eval; "Multi-agent systems excel at valuable tasks that involve heavy parallelization, information that exceeds single context windows, and interfacing with numerous complex tools." โ€” https://www.anthropic.com/engineering/multi-agent-research-system +- **The reconciliation both sides state:** Anthropic concedes "most coding tasks involve fewer truly parallelizable tasks than research." Cognition's product is a *coding* agent. LangChain independently landed on the same boundary in their own reference implementation: "**We restrict multi-agent to research, and write the report in one-shot**" โ€” https://www.langchain.com/blog/open-deep-research โ€” 2025-07-16. Anthropic's Nov 2025 long-running-agents post is agnostic and admits uncertainty about whether specialized sub-agents help. + +**D3 โ€” "More compute = better research" vs cost discipline.** +Anthropic asserts token usage explains **80%** of eval variance and accepts **15ร—** chat token cost. Microsoft's Azure docs push the opposite: lower reasoning effort, consolidate indexes to reduce fan-out, reorganize content "so the most relevant information can be found with fewer sources." OpenAI's own GPT-5 prompting guidance says "**Prefer acting over more searching**" and shows a 2-tool-call budget example. Exa monetizes the middle by exposing compute as a per-request price tier. Perplexity's cost breakdown shows reasoning tokens outweighing output tokens ~6ร—. + +**D4 โ€” Where the stopping rule lives.** OpenAI = developer hard cap. Anthropic = model judgment + prompt heuristics + `max_uses` backstop. Microsoft Researcher = marginal-information threshold. Google = trained dead-end detection. Azure agentic retrieval = **no stopping rule at all** (single fixed fan-out). Exa = a monetary cap. + +**D5 โ€” RRF vs weighted score fusion.** Weaviate moved its default *off* RRF; Qdrant calls RRF "the de facto standard"; MongoDB GA'd both and declines to steer; Pinecone claims reranking beats both by 8% on BEIR. **Nobody has published a head-to-head with numbers.** + +**D6 โ€” Late interaction: production or not.** Vespa "Yes" explicitly; Weaviate GA with a candid recall cliff; Elastic GA per release blog; Qdrant reranker-only and client-library-only; Pinecone no native support, calls end-to-end ColBERT "notably slow"; Vectara actively rejecting patch-level multi-vector. + +**D7 โ€” LLM rerankers.** Voyage publishes that they are 25โ€“60x more expensive, up to 48x slower, and *degrade* NDCG@10 when the first stage is good. Meanwhile Chroma trains a 20B LLM to *be* the retriever, SID/turbopuffer trains an LLM search agent claiming 0.77 vs 0.45 recall over classical rerank pipelines (5.5s vs 131s, $0.62 vs $240 per 1k questions), and Exa sells frontier-LLM agentic search at 90% Deep Search QA. **The two camps are not measuring the same thing** โ€” Voyage measures *reranking a fixed candidate list*; the others measure *multi-hop search where the agent issues new queries*. That distinction is doing all the work and no vendor states it plainly. + +**D8 โ€” Chunk size.** Chroma's own eval says ~200 tokens dominates and calls 800-token defaults worst-in-class. OpenAI's default is 800/400. Elastic's `semantic_text` default is "250 words (approximately 400 tokens)." Jina argues the question is wrong-framed and you should late-chunk a long-context encoder instead. + +**D9 โ€” Contextual retrieval, publicly disparaged.** Jina calls Anthropic's method "a brute-force approach" and claims late chunking is cheaper, faster, and more robust to bad boundaries. No matched head-to-head exists. No vendor ships Anthropic-style contextual chunk augmentation as a GA feature. + +**D10 โ€” GraphRAG vs vector-only vs no-index.** Microsoft Research argues graph structure is required for global/aggregative questions, then walks back the cost 1000x with LazyGraphRAG. Anthropic argues the opposite for code: no index at all. Glean and Perplexity both land on hybrid. Cohere frames it as "From GraphRAG to agentic search" (https://cohere.com/blog/ai-retrieval-graphrag-and-agentic-search โ€” 2025-04-28; body would not render). + +**D11 โ€” Prefab eval metrics: a flat "No" vs an entire industry shipping them.** Hamel & Shreya: "No." / "All you get from using these prefab evals is you don't know what they actually do." Against: Ragas, Azure Foundry (11 agent evaluators), MLflow (14+ predefined judges), Galileo, Phoenix, Braintrust autoevals. **Partial reconciliation:** MLflow concedes the point quantitatively โ€” aligned judges cut FP/FN "30-50%" vs generic prompts. Anthropic splits the difference: "You don't need to invent an evaluation from scratchโ€ฆ Use these methods as a foundation, then extend them to your domain." + +**D12 โ€” Binary vs Likert.** Hamel: 1โ€“5 scales mean "you're doing it wrong." Eugene Yan: "where possible, I have my evaluators return binary outputs." Databricks: yes/no. Azure Foundry: binary output โ€” **but internally 1โ€“5 with a threshold** for several evaluators, and Customer Satisfaction is straight Likert. Contextual AI's LMUnit is explicitly continuous 1โ€“5. Net: the field converged on **binary at the decision boundary** while several vendors keep a graded score underneath. + +**D13 โ€” Reference-free metrics.** Ragas and LlamaIndex both advertise that many modules "do *not* require ground-truth labels." Hamel, Yan, Bergum, and MLflow all argue a judge is meaningless until validated against human labels (TPR/TNR, Cohen's ฮบ, โ‰ฅ10โ€“20 labeled examples). Bergum's calibrate-small-then-scale method is the practical middle path. + +**D14 โ€” Trajectory vs end-state agent eval.** Azure Foundry endorses process evaluation (5 tool-level evaluators). Anthropic explicitly rejected turn-by-turn for their research agent because "agents could follow different valid paths." LangSmith's `subset`/`superset`/`unordered` modes are essentially a compromise for exactly this. + +**D15 โ€” LLM-judge trust levels.** Anthropic is the most conservative shipped position ("we do not take eval scores at face value until someoneโ€ฆ reads some transcripts"). Eugene Yan's survey supplies the floor: HaluEval best model 58.5% accuracy; 30โ€“60% recall on inconsistency detection despite >95% specificity. Elastic reports ~80% LLMโ€“human agreement on relevance; Bergum reports near-total agreement within one level. **Net: judges are reliable enough for relevance labeling, unreliable for hallucination detection โ€” and vendors ship groundedness judges anyway.** + +**D16 โ€” Where the LLM belongs in relevance tuning (a vendor publishing evidence against its own pitch).** Vespa's autoresearch experiment: a free-form LLM agent got the biggest in-domain lift (+8.9%) but retained only **21%** of it out-of-domain; the same agent constrained to Vespa's rank-feature library retained **99%**. Manual tuning retained 80%. + +**D17 โ€” Priority on "code execution instead of tool calls."** Cloudflare published 2025-09-26; Anthropic published 2025-11-04 without citing it. Different rationales: Cloudflare = models write TypeScript better than they emit tool-call syntax (a training-data argument); Anthropic = the filesystem enables on-demand tool-definition loading and keeps intermediate data out of context. + +--- + +## SECTION 7 โ€” SOURCES REJECTED FOR QUALITY + +- `connorshorten300.medium.com/muvera-with-rajesh-jayaram` โ€” Medium; fails the bar even though the author is Weaviate staff. +- Medium posts by Jagadeesan Ganesh, Nikhil Mogre, Abdullah Grewal/buzzgrewal, HARSHA J S, DhanushKumar/Stackademic, Towards AI โ€” anonymous/individual Medium, explicitly out of scope. +- `towardsdatascience.com` โ€” "How Cursor Actually Indexes Your Codebase" (third-party reverse-engineering); Vespa ODQA repost (aggregator, 2020-era). +- `letsdatascience.com/blog/vector-databases-compared-...` โ€” SEO listicle, unknown author. +- `cipherprojects.com/.../weaviate-vs-qdrant-...` โ€” vendor-adjacent marketing roundup, unknown author. +- `chat-deep.ai/docs/deepseek-vector-database-guide/` โ€” content-farm aggregator. +- `zilliz.com/comparison/weaviate-vs-qdrant` โ€” first-party domain but competitive-marketing page, not engineering content. +- `interestingengineering.substack.com/p/from-bm25-to-agentic-rag` โ€” third-party newsletter. +- KDnuggets "How to Implement Agentic RAG Using LangChain" Parts 1โ€“2 โ€” third-party tutorial site. +- `marktechpost.com`, `pureai.com`, `hyper.ai` (FRAMES writeups) โ€” SEO news aggregators / mirrors. +- `llm-stats.com/benchmarks/frames` โ€” third-party leaderboard scraper. +- `siftq.com/blog/using-frames-benchmark-...` โ€” vendor SEO blog, not on the list. +- `leeroopedia.com` (Ragas agent eval) โ€” auto-generated wiki, unattributed. +- `aws.amazon.com/blogs/machine-learning` (Bedrock + Ragas) โ€” AWS not on the allowed-org list; third-party framing of Ragas. +- `elastic.co` "Agentic RAG with LangChain & Elasticsearch" โ€” not first-party LangChain (used for the LangChain claim it was sought for). +- NVIDIA "Traditional RAG vs. Agentic RAG" โ€” vendor, but not on the allowed list for that claim. +- `forum.weaviate.io` threads, `news.ycombinator.com/item?id=44387617`, `community.databricks.com/t5/technical-blog/...` โ€” community/forum content, not official. +- `github.com/mlbrnm/contextualretrieval`, `github.com/autollama/autollama` โ€” community implementations, not vendor-published. +- `github.com/explodinggradients/ragas` issue #2122 โ€” a user bug report, not documentation. +- `podcasts.chainofthought.xyz/.../jo-kristian-bergum` โ€” third-party podcast summary, not Bergum's own writing. +- Latent Space "Normsky architecture" podcast โ€” third-party podcast (the turbopuffer Latent Space post *was* used because it is hosted on turbopuffer.com as first-party). +- `wowelec.wordpress.com`, `robertheubanks.substack.com` โ€” personal blogs, not named practitioners on the list. +- `newsletter.weaviate.io/p/muvera-...` โ€” newsletter digest; superseded by the primary blog post. +- `news.microsoft.com/source/features/ai/6-surprising-ways-...` โ€” consumer PR feature, no orchestration detail. +- `comet-framer-prod.perplexity.ai` โ€” staging host, not a citable published page. +- `ZenML` LLMOps Database entry on Cursor โ€” secondhand summary; used the Cursor primary instead. +- Educative.io, DataCamp, IBM Think, perfectiongeeks.com, langchain-opentutorial.gitbook.io, buildmvpfast.com, digitalapplied.com, mindstudio.ai, meta-intelligence.tech, codemyspec.com, developersdigest.tech, matthewkruczek.ai, usewire.io, mcp.directory, theunwindai.com โ€” third-party course/SEO/affiliate content. +- X/Twitter and LinkedIn posts (Jerry Liu, Cursor, Cheng Lou, jobergum) โ€” surfaced in search, not fetchable as stable dated content; treated as unverified. +- ~30 arXiv preprints on agentic RAG surveys, benchmarks, and reference architectures (2501.09136, 2504.13587, 2507.09477, 2604.16394, etc.) โ€” academic work outside the stated vendor/practitioner bar. Exceptions made only where the paper *is* the org's first-party publication (OpenAI BrowseComp, Google FRAMES, MMTEB). +- `ir.nist.gov/trec-covid` PDFs, `github.com/castorini/anserini` โ€” IR toolkit/run files, not vendor architecture publications. + +--- + +## SECTION 8 โ€” GAPS AND UNVERIFIED ITEMS (read before publishing anything from this) + +**403 / unfetchable first-party pages (findings rest on snippets or substitutes):** +- `openai.com/index/*` (introducing-deep-research, browsecomp, new-tools-for-building-agents, introducing-company-knowledge, memory-and-new-controls-for-chatgpt) โ€” all 403. **The widely-cited Deep Research HLE 26.6% / GAIA figures are NOT verified here and are not asserted.** ChatGPT consumer memory is entirely unverified. +- `help.openai.com/*` โ€” 403. The company-knowledge finding is snippet-based. +- `perplexity.ai/hub/blog/*` โ€” 403 across the board. All Perplexity findings come from `research.perplexity.ai` and `docs.perplexity.ai`, which do render. +- `cdn.openai.com/.../a-practical-guide-to-building-agents.pdf` โ€” returned unparseable binary (7MB). **OpenAI's single-vs-multi-agent guidance is unverified.** +- `techcommunity.microsoft.com/.../researcher-agent-...` โ€” renders title only. The ฮ”I<ฮต stopping rule is snippet-sourced. +- `contextual.ai/new/groundedness-scoring-...`, `learn.microsoft.com/.../groundedness`, Google `dynamicRetrievalConfig` blog, Google FRAMES, MMTEB paper โ€” snippet-only. +- `cohere.com/blog/ai-retrieval-graphrag-and-agentic-search` and `cohere.com/blog/rerank-4` โ€” nav/header only. **Cohere North orchestration and Rerank v4.0 benchmarks are genuine gaps.** Cohere Rerank 3.5's date is contradictory (page says 2024-12-02, snippet said 2025-07-10). +- 404s (URL guesses, do not cite): `developers.openai.com/blog/agentkit/`, `docs.cloud.google.com/gemini-enterprise/docs/overview`, `redis.io/blog/`, `docs.letta.com/memory/memfs`, `devin.ai/blog/why-we-built-riptide`, `devin.ai/blog/codemaps`, several `learn.microsoft.com/agent-framework/.../memory` paths, `docs.cohere.com/docs/rerank-best-practices`, W&B Weave scorers (403), Arize Phoenix pre-tested-evals tables (404). + +**Uncovered orgs:** Meta AI (no first-party agentic-retrieval product documentation found). Cohere North. W&B Weave. Patronus AI. Redis. Zilliz/Milvus (page rendered title-only; RRF/WeightedRanker support is snippet-level only). Databricks long-context-RAG posts exist but were not fetched โ€” **do not cite Databricks long-context numbers from this report** (the Instructed Retriever numbers in ยง1.4 *were* fetched and are solid). + +**Status not stated on source page:** Vertex AI Agent Engine Memory Bank GA-vs-preview. Google check-grounding GA-vs-preview (inferred GA from context). + +**Under-searched, not proven absent:** HyDE as a shipped vendor feature. Provence-style context pruning as a vendor feature (only Chroma Context-1 and Weaviate guidance verified). Windsurf's original retrieval posts are gone post-acquisition (404), so their position is only reachable through Cognition's SWE-grep/SWE-1.5 posts. + +**Budget note:** the session's 200 WebSearch calls were exhausted mid-research. Late-stage sources were reached by direct URL fetch. Any finding marked `[snippet-only]` should be re-verified before publication. \ No newline at end of file diff --git a/Documentation/research_roadmap.md b/Documentation/research_roadmap.md new file mode 100644 index 00000000..732814c3 --- /dev/null +++ b/Documentation/research_roadmap.md @@ -0,0 +1,148 @@ +# Evidence-Based Roadmap + +_Created 2026-08-08. Status of every item here is **PLANNED, not implemented**, +except where a row says otherwise โ€” when an item ships, move its row to +`improvement_plan.md`'s completed table and update the component docs. Do not +describe any of this as current behavior._ + +**Phase 1 is complete (2026-08-09).** Items 1.1, 1.2 and 1.3 were measured and +decided; the adopted defaults, the joint matrix behind them and what stays +opt-in are in [`../eval/DECISIONS.md`](../eval/DECISIONS.md). + +This roadmap turns the findings in [`research/`](research/) into staged, +testable changes. Ordering principle, straight from the evidence: **retriever +and reranker quality dominate everything else** (+14.2 pts from an embedder +swap on BrowseComp-Plus, ACL 2026; +17.2 pp MRR@3 from a cross-encoder), every +clever layer on top is conditional, and no upgrade counts until it is measured +on our own eval set (public-leaderboard deltas under ~2 points do not transfer; +ranking inversions on production corpora are documented). + +--- + +## Phase 0 โ€” Evaluation harness (prerequisite; nothing else lands without it) + +The single most consistent practitioner finding: generic metrics create false +confidence; teams need a small, owned eval set and binary judgments +(Hamel/Husainโ€“Shankar FAQ; Anthropic "20โ€“50 tasks from real failures is a +great start"; Chroma's generative-benchmarking ranking inversions). + +| # | Item | Detail | Acceptance | +|---|------|--------|------------| +| 0.1 | Gold retrieval set | Reverse-generate ~50โ€“100 queries from indexed chunks across 2โ€“3 real corpora + the planted-fact test PDF, using structured dimension tuples (topic ร— question-type ร— difficulty), not freeform "give me questions". Store under `eval/goldset/` with the chunk IDs that must be retrieved. | Set committed; every query hand-checked once | +| 0.2 | Retrieval metrics runner | `eval/run_eval.py` hits the RAG API: **recall@k** at first stage, **nDCG@10 after rerank** โ€” the two metrics the evidence says matter. Runs in minutes, no LLM judge needed. | Baseline numbers for the current stack recorded in `eval/BASELINE.md` | +| 0.3 | Groundedness judge | Binary pass/fail LLM judge for answer faithfulness, validated against ~20 hand-labeled answers (report TPR/TNR, not raw agreement). Binary, not Likert, per the near-unanimous practitioner canon. | Judge agrees with hand labels โ‰ฅ90% before it gates anything | +| 0.4 | End-to-end smoke | Scripted: index the test PDF โ†’ 5 planted-fact questions โ†’ assert grounding + citation presence + persistence round-trip. Extends the E2E script already exercised on 2026-08-08. | Runs green on the current tree | + +**Effort:** ~1 day. **Risk:** low. **Everything below cites Phase 0 numbers in +its acceptance criteria.** + +## Phase 1 โ€” Component upgrades (A/B against Phase 0, adopt only on wins) + +| # | Item | Evidence | Plan | Risk | +|---|------|----------|------|------| +| 1.1 โœ… **DONE 2026-08-09** โ€” reranking now ships **off**, with `Qwen/Qwen3-Reranker-4B` as the model the toggle loads. _(Superseded 2026-08-14, arm G: reranking now ships **on** by default with `min_score: 0.5` / `min_keep: 3` threshold selection.)_ Decision + numbers: [`../eval/DECISIONS.md`](../eval/DECISIONS.md), evidence: [`../eval/decisions/reranker.md`](../eval/decisions/reranker.md) | **Reranker A/B: Qwen3-Reranker-4B (and 0.6B) vs bge-reranker-v2-m3** | bge-v2 lineage static for 2 years; near-zero on instruction-following (FollowIR โˆ’0.01) and weak on code (41.38 MTEB-Code); Qwen3-Reranker-4B leads permissive options (69.76 MTEB-R, Apache 2.0). Reranker is the highest-ROI slot in the stack. | Config-swap via `RERANKER_MODEL`; verify the `rerankers` library loads Qwen3's causal yes/no-logit scoring (it is NOT a SequenceClassification model โ€” may need a small custom scorer). Measure nDCG@10 delta and per-query latency on MPS. | Integration: medium. Throughput regression is expected; accept only if quality delta justifies it, else keep 0.6B or bge | +| 1.2 โœ… **DONE 2026-08-09** โ€” `microsoft/harrier-oss-v1-0.6b` adopted as the default with the query prefix on, in the same re-index window as 1.1 and the two index-format fixes. Decision + numbers: [`../eval/DECISIONS.md`](../eval/DECISIONS.md), evidence: [`../eval/decisions/embedder.md`](../eval/decisions/embedder.md) | **Embedder audit + A/B: instruction prefix, then harrier-oss-v1 vs Qwen3-Embedding** | All top-2026 embedders are instruction-tuned (+1โ€“5% from the query-side prefix alone); microsoft/harrier-oss-v1 (MIT, Mar 2026) beats Qwen3-Embedding at equal size (0.6B: 69.0 vs 64.3 MMTEB). | First verify our query path sends the `Instruct: โ€ฆ Query: โ€ฆ` prefix (free win if missing). Then A/B harrier-0.6b and Qwen3-4B on the gold set. **Changing the default forces re-indexing** โ€” decide once, alongside 1.1, in a single index-format window. | Low code risk; migration cost is the re-index. harrier is decoder/last-token-pooling like Qwen3, so `embedders.py` should need at most a config entry | +| 1.3 | **GLM-OCR behind Docling for scanned/complex PDFs** โ€” _spike complete 2026-08-09: **GO-LATER**, see `eval/decisions/glm-ocr-spike.md`_ | 0.9B MIT specialist VLM; **95.22 OmniDocBench v1.6_full (third, behind PaddleOCR-VL-1.6 96.34 and MinerU2.5-Pro 95.75)** โ€” an earlier "#1 / beats GPT-5.2 by ~10 pts" framing from the research sweep did not survive source verification and is retracted. Spike proved the serving path on Apple Silicon: official `ollama pull glm-ocr` (2.2GB), and our pinned docling 2.118.1 already ships a `glm_ocr` preset with an Ollama override. Parse quality on a degraded scanned invoice: 30/30 table cells vs the current chain losing every price. | Blocked on three items before adoption: (a) Ollama's Modelfile ignores prompts, so GLM-OCR's table/formula modes are unreachable there; (b) deterministic double-transcription on some pages; (c) docling flattens its pipe tables (`tables: 0`). Prerequisite: add a scanned/tabular corpus with ground truth to `eval/` โ€” nothing in Phase 0 exercises the OCR branch โ€” and A/B GLM-OCR against docling's other 2026 presets (`lightonocr`, `dots_ocr`, `nanonets_ocr2`โ€ฆ), not as a foregone winner. | Cheaper wins shipped first: the RapidOCR probe bug is fixed (stale module name sent every scan to Tesseract CLI), and `pip install ocrmac` gives macOS the best classic chain | + +**Effort:** ~2โ€“3 days including eval runs. **Decision gate:** adopt each only +on a measured win; record adopted/rejected + numbers in `eval/DECISIONS.md`. +**Gate cleared 2026-08-09** โ€” see [`../eval/DECISIONS.md`](../eval/DECISIONS.md). +1.1 and 1.2 could not be decided separately (the reranker's value depends on the +first stage), so they were re-measured jointly and shipped in one window +together with the embedder-identity guard and cosine normalization that the 1.2 +audit flagged. 1.3 remains GO-LATER: no code was written. + +## Phase 2 โ€” Pipeline-shape fixes โ€” _COMPLETE 2026-08-09; measurements in `eval/decisions/phase2-pipeline.md` and `phase2-gateway.md`; Landed rows graduated to improvement_plan.md. Notable: 2.2 measured negative on truly-decomposing queries and ships disabled; 2.1's suggested signal was replaced after measuring anti-correlation. (2.2's "ships disabled" posture was superseded: arm G, 2026-08-14, turned the reranker on by default so rerank-decomposition now applies, and arm H, 2026-08-15, made the pooled first stage the default decomposition path.)_ + +| # | Item | Evidence | Plan | +|---|------|----------|------| +| 2.1 | **Evidence-sufficiency retry** | One conditional second retrieval iteration captures ~95% of deep-loop gains (+7.1 EM for iteration 1โ†’2 on a local 7B; iterations โ‰ฅ3 are noise). Terminate on evidence sufficiency, not query count. | If top-k rerank scores fall below a threshold, reformulate once (enrichment model) and retrieve again; hard cap 2 iterations; off in `fast` profile. Surface as a step in the SSE cascade so the UI shows it | +| 2.2 | **Decomposition at rerank, not first-stage** | Decomposition at initial retrieval "frequently harms via semantic dilution"; it helps applied at reranking (2026 finding, MultiConIR/SSRB). | Keep the full query for hybrid retrieval; score candidates against sub-queries during rerank. Touches `retrieval_pipeline.py` only | +| 2.3 | **Cheapen gateway routing** | Pre-retrieval LLM routing is the weakest pattern in the 2026 evidence (four ML approaches failed; TF-IDF+SVM โ‰ฅ LLM routers at ~zero cost; fixed-hybrid beat adaptive routing). | Replace the gateway's LLM routing call with a cheap gate (heuristics + optional logit-margin TARG-style check), keep `force_rag`, keep agent-side triage as the single LLM routing layer. Net: one less LLM call per message, simpler failure surface | +| 2.4 | **Dedicated verifier model (optional)** | Verification helps only as an external check; a 4-bit 1B verifier (ThinknCheck, CC0) now beats the 7B 2024 SOTA; Granite Guardian (Apache) is the permissive alternative. Current LLM-prompt verifier works but is slower and uncalibrated. | Add `VERIFIER_MODEL` option: local NLI/verifier model scoring answer-vs-evidence per sentence; keep the LLM prompt as fallback. Present `[Confidence: N%]` as UX, never as a calibrated measurement (document this) | +| 2.5 | **Delete the graph module** | GraphRAG loses single-hop, contested multi-hop gains, 41โ€“57ร— indexing / up to ~377ร— query-token cost; unreachable in this repo already (no profile sets `graph_strategy`). | Remove `graph_extractor.py`, `GraphRetriever`, `GraphQueryTranslator` + config remnants; note the decision and evidence in `design_rationale.md` (resolves improvement_plan ยง9) | + +**Effort:** ~2 days. Each item is independently shippable and independently +revertible; each lands with a Phase-0 eval delta. + +## Phase 3 โ€” Documentation: make the evidence part of the repo's argument + +**3.1 is complete (2026-08-09):** [`design_rationale.md`](design_rationale.md) +ships one section per component with its evidence and its eval number, plus the +"deliberately not implemented" list. 3.2's README/roadmap/docs-index cross-links +landed with it. + +| # | Item | Plan | +|---|------|------| +| 3.1 โœ… **DONE 2026-08-09** โ†’ [`design_rationale.md`](design_rationale.md) | One section per component: what localGPT does, and the evidence for why (with citations into `research/`). Includes a **"deliberately not implemented"** list โ€” HyDE-by-default, multi-query expansion, weighted fusion knobs, GraphRAG, vendor memory systems, deep subagent fan-out โ€” each with its citation, so future contributors don't re-add deprecated patterns | +| 3.2 | Cross-link | README gets one line pointing at the rationale; `improvement_plan.md` links each open item to its evidence; this roadmap's rows move to improvement_plan as they complete | +| 3.3 | Honesty rule (already in force) | Every phase lands code + docs + eval delta in the same change. Nothing in `design_rationale.md` may describe unshipped behavior โ€” that is what this roadmap file is for | + +**Effort:** ~half a day, mostly distillation from the synthesis already written. + +## Explicitly out of scope (evidence-negative โ€” revisit only with new evidence) + +- **Deep agent loops / parallel subagent fan-out** โ€” pays on open-web breadth-first research at ~15ร— token budgets; fails silently elsewhere (41.8% of failures at the hand-off). +- **RL-trained searchers** โ€” no out-of-distribution transfer (Search-R1 = its base model, ACL 2026); also mis-calibrates confidence, which would break 2.3's logit gate. +- **Always-on HyDE / multi-query** โ€” measured negative on entity/numeric corpora; multi-query scored below plain BM25. +- **Vendor memory layers** โ€” the null baseline (RAG over the transcript) wins; our session store already is the null baseline. +- **Token-level context compression (LLMLingua-style)** โ€” โ‰ค18% best-case speedup, net-negative outside a narrow window; we already ship the surviving alternative (Provence pruning). + +## Sequencing summary + +``` +Phase 0 (eval harness) โ”€โ”€โ–บ Phase 1 (reranker / embedder / parser A/Bs) โ”€โ”€โ–บ Phase 2 (pipeline shape) โ”€โ”€โ–บ Phase 3 (rationale docs) + ~1 day ~2โ€“3 days ~2 days ~0.5 day + โœ… done โœ… done 2026-08-09 (1.3 = GO-LATER) +``` + +Phase 0 blocks everything. Phases 1 and 2 can interleave per-item, but 1.1/1.2 +should conclude before any re-index-requiring release. Phase 3.1 can be drafted +in parallel at any time. + +## Phase 4 โ€” Ideas adopted from agentic-file-search (implemented + benchmarked 2026-08-09) + +**Phase 4 is complete.** Every item below was implemented flag-gated, benchmarked +on/off, and decided at the adoption gate. Verdicts (evidence in +`eval/decisions/phase4-*.md`): + +| # | Verdict | One-line reason | +|---|---------|-----------------| +| 4.1 | **REJECTED as default** โ€” implemented, off (was HOLD) | The HOLD's condition was executed 2026-08-12: with the num_ctx fix in place the lift did not survive โ€” the escalation-off baseline on the identical fire subset went 0/9โ†’7/9 (the "lift" was front-truncation favoring the tail-appended document) and both product-default fires were regressions. Flag and code kept. `eval/decisions/phase4-escalation-rerun.md`. | +| 4.2 | **Extraction ADOPTED (on); hop REJECTED as a default (off, flag kept)** | Extraction is free, regex-only and index-inert. The hop fires 0/11 at the shipped k=20, hits 0/11 expected sources where forced (target selection is query-blind), and raising `k` beats it at equal context budget in 3 of 4 cells. | +| 4.3 | **boost HOLD (off); restrict REJECTED** | boost: nDCG@10 0.773โ†’0.879 on the heterogeneous slice but a loss on `mixed` โ€” per-index opt-in candidate, not a default. restrict removed the answer document entirely for 4 queries per corpus (recall@20 1โ†’0). | +| 4.4 | **ADOPTED** (no flag: inert without a `filters` argument) | Byte-identical behavior when unused (md5-verified); injection refused, both search legs prefiltered. **Correction to the row below: page/date filters did NOT ship** โ€” they live inside the metadata JSON string and need real columns + a re-index. | +| 4.5 | **ADOPTED (on by default)** | Zero-risk observability; `token_usage` per stage on `/chat` and the SSE `complete` event. | +| 4.6 | **ADOPTED** (opt-in CLI) | `python -m rag_system.main ask ""`; ephemeral index, verified cleanup including on SIGTERM. | + +Original planning table follows, kept for the evidence trail: + +Source: [PromtEngineer/agentic-file-search](https://github.com/PromtEngineer/agentic-file-search) +(FsExplorer lineage โ€” an agentic filesystem QA agent: three-phase scan/dive/backtrack, +Docling parsing, grep/glob tools, DuckDB+VSS indexed search with a metadata-filter DSL, +exploration traces, per-query token/cost tracking). The evidence lens for every item: +escalate-don't-pre-decide (PEA-CAE), retriever quality dominates agency +(BrowseComp-Plus), and filesystem agents win small corpora but lose to ranked +retrieval at scale with ~39x token cost (BM25-wins-at-scale) โ€” so we adopt its +*escalation mechanisms*, not its loop. + +| # | Item | From their design | Our incorporation | Evidence fit | +|---|------|-------------------|-------------------|--------------| +| 4.1 | **Full-document escalation** | `parse_file` / `get_document` deep-read tools | When the evidence-sufficiency retry (2.1) still lands weak, reassemble the top-cited document IN ORDER from its chunks (chunk_index exists in metadata) and hand it to synthesis, token-capped, one document max. New RAG API helper + agent step; SSE event `document_escalation`. | PEA-CAE escalation; DOS-RAG document-order finding; bounded, not a loop | +| 4.2 | **Cross-reference hop** | Phase-3 backtracking on "See Exhibit B" | Index-time regex extraction of intra-corpus references (exhibit/section/filename mentions) into chunk metadata; query-time one-hop pull of the referenced doc's overview + top chunks when a top-ranked chunk carries one. Capped at one hop, no LLM in the hop. | Fixes the real "cross-references are invisible to embeddings" gap without unbounded agency | +| 4.3 | **Overview prefilter ("peripheral vision")** | Phase-1 parallel scan + RELEVANT/MAYBE/SKIP triage | We already build per-doc overviews; embed them once and use overview-vs-query similarity to boost/restrict chunk retrieval to top documents on multi-document indexes. No per-query LLM cost (their scan phase is an LLM call per document โ€” the measured 39x failure mode). | Jason Liu facets/peripheral-vision; avoids their linear-cost scan | +| 4.4 | **Metadata filter DSL / self-query** | `semantic_search(filters="field=value, field in (a,b)")` on DuckDB | Same surface on LanceDB `where` clauses: accept `filters` on /chat + /chat/stream, wire to chunk metadata (document name, page, date). LLM filter-extraction later โ€” small local models measured ~0.999 F1 on easy/medium filter translation. | The one query-planning technique local models already nail (component-map ยง6.4) | +| 4.5 | **Per-query token/cost tracking** | TokenTracker + cost summary per query | Surface Ollama's prompt_eval_count/eval_count per stage in the SSE `complete` event and UI (local cost = time + watts, still worth showing). | Their nicest UX idea; zero risk | +| 4.6 | **Ephemeral "ask a folder" mode** | Index-free filesystem QA | `python -m rag_system.main ask ""`: build a temp in-memory/throwaway index (fast profile, no enrichment), answer, delete. Same pipeline, no agent loop. | FS-agents win small corpora โ€” but an ephemeral *index* beats an ephemeral *agent* on our own retriever-dominance evidence | + +**Deliberately NOT adopted** (goes in design_rationale ยง13 when implemented): the +LLM-per-document scan phase (linear cost in corpus size โ€” the exact pattern +BM25-wins-at-scale measured at ~39x tokens), the free-form ReAct exploration loop +(search volume correlates weakly with quality; silent hand-off failures), cloud +Gemini (violates the local/privacy premise), llama-index-workflows + DuckDB +(duplicate our framework-free stack and LanceDB). + +Sequencing: 4.5 and 4.4 are independent quick wins; 4.1 depends on 2.1's signal +(shipped); 4.2/4.3 are index-format-adjacent and should share a re-index window; +4.6 is CLI-only. All gated on the Phase 0 harness like everything else โ€” 4.2 and +4.3 need multi-document gold queries with cross-references added to eval/goldset. diff --git a/Documentation/retrieval_pipeline.md b/Documentation/retrieval_pipeline.md index d4ae7168..d74a74b5 100644 --- a/Documentation/retrieval_pipeline.md +++ b/Documentation/retrieval_pipeline.md @@ -1,616 +1,356 @@ # ๐Ÿ“ฅ Retrieval Pipeline -_Maps to `rag_system/pipelines/retrieval_pipeline.py` and helpers in `retrieval/`, `rerankers/`._ +_Maps to `rag_system/pipelines/retrieval_pipeline.py`, orchestrated by `rag_system/agent/loop.py`, with helpers in `retrieval/` and `rerankers/`._ ## Role -Given a **user query** and one or more indexed tables, retrieve the most relevant text chunks and synthesise an answer. +Given a user query and one LanceDB table, retrieve the most relevant chunks and synthesise an answer with its source documents. + +Two objects share the work: + +* **`Agent`** (`rag_system/agent/loop.py`) owns triage, the semantic cache, query decomposition, sub-answer composition, conversation history and verification. +* **`RetrievalPipeline`** (`rag_system/pipelines/retrieval_pipeline.py`) owns everything from retrieval to synthesis for a *single* query string. `Agent` may call it once, or once per sub-query in parallel. ## Sub-components -| Stage | Module | Key Classes / Fns | Notes | -|-------|--------|-------------------|-------| -| Query Pre-processing | `retrieval/query_transformer.py` | `QueryTransformer`, `HyDEGenerator`, `GraphQueryTranslator` | Expands, rewrites, or translates the raw query. | -| Retrieval | `retrieval/retrievers.py` | `BM25Retriever`, `DenseRetriever`, `HybridRetriever` | Abstract over LanceDB vector + FTS search. | -| Reranking | `rerankers/reranker.py` | `ColBERTSmall`, fallback `bge-reranker` | Optionally improves result ordering. | -| Synthesis | `pipelines/retrieval_pipeline.py` | `_synthesize_final_answer()` | Calls LLM with evidence snippets. | -## End-to-End Flow +| Stage | Module | Key classes / functions | Notes | +|-------|--------|-------------------------|-------| +| Query decomposition | `retrieval/query_transformer.py` | `QueryDecomposer.decompose()` | Optional. Splits a query into standalone sub-queries; runs on the utility model. | +| Retrieval | `retrieval/retrievers.py` | `MultiVectorRetriever.retrieve()` | Runs LanceDB full-text and/or vector search over one table. There is no separate BM25 retriever class โ€” lexical search is LanceDB's native FTS index. | +| Reranking | `pipelines/retrieval_pipeline.py`, `rerankers/reranker.py` | `_get_ai_reranker()`, `QwenRerankerScorer`, `rerankers.Reranker`, `CrossEncoderReranker` | On by default since arm G (2026-08-14); the model loads lazily on the first reranked query. Qwen3-Reranker names go to `QwenRerankerScorer`; otherwise the model is loaded through the `rerankers` library. `CrossEncoderReranker` is the non-library fallback branch. | +| Sentence pruning | `rerankers/sentence_pruner.py` | `SentencePruner.prune_documents()` | Provence (`naver/provence-reranker-debertav3-v1`). Off unless requested. | +| Synthesis | `pipelines/retrieval_pipeline.py` | `_synthesize_final_answer()` | Streams an LLM completion on the generation model. | +| Verification | `agent/verifier.py` | `Verifier.verify_async()` | See `verifier.md`. | + +### Removed: the graph path + +`GraphQueryTranslator`, `GraphRetriever` and `GraphExtractor` were **deleted on +2026-08-09** (roadmap item 2.5). They were unreachable โ€” no shipped profile ever +set `graph_strategy` โ€” and the evidence is against re-adding them: GraphRAG +*loses* on single-hop retrieval, its multi-hop gains range from +3 points to +27 +depending on how well the vector baseline is tuned, and it costs **41โ€“57ร— at +indexing** and up to **~377ร— in query tokens**. See +[`research/academic-evidence-2026.md`](research/academic-evidence-2026.md) ยง6. +`networkx` and `fuzzywuzzy` left `requirements.txt` with them. + +## End-to-end flow ```mermaid -flowchart LR - Q["User Query"] --> XT["Query Transformer"] - XT -->|variants| RETRIEVE - subgraph Retrieval - RET_BM25[BM25] --> MERGE - RET_DENSE[Dense Vector] --> MERGE - style RET_BM25 fill:#444,stroke:#ccc,color:#fff - style RET_DENSE fill:#444,stroke:#ccc,color:#fff +flowchart TD + Q["User query"] --> T["Triage (see triage_system.md)"] + T -- direct_answer --> DA["Direct LLM answer"] + T -- rag_query --> C{"Semantic cache hit?"} + C -- yes --> OUT["answer + source_documents"] + C -- no --> D{"Decomposition enabled?"} + D -- yes --> SUB["1..N sub-queries retrieved in parallel,
candidates pooled + deduped (arm H default)"] + D -- no --> ONE["Single query"] + SUB --> RP + ONE --> RP + subgraph RP [RetrievalPipeline.run] + R1["Retrieve (hybrid RRF / vector_only / fts_only)"] --> R2["Late-chunk table + sibling merge (optional)"] + R2 --> R3["AI rerank (on by default)"] + R3 --> R4["Context expansion (optional)"] + R4 --> R5["Provence pruning (optional)"] + R5 --> R6["Synthesis (streamed)"] end - MERGE --> RERANK - RERANK --> K[["Top-K Chunks"]] - K --> SYNTH["Answer Synthesiser\n(LLM)"] - SYNTH --> A["Answer + Sources"] + RP --> V["Verification (optional)"] + RP -. "opt-in compose path" .-> COMP["Compose sub-answers"] + COMP --> V + DA -- "no source_documents, skipped" --> V + V --> OUT ``` -### Narrative -1. **Query Transformer** may expand the query (keyword list, HyDE doc, KG translation) depending on `searchType`. -2. **Retrievers** execute BM25 and/or dense similarity against LanceDB. Combination controlled by `retrievalMode` and `denseWeight`. -3. **Reranker** (if `aiRerank=true` or hybrid search) scores snippets; top `rerankerTopK` chosen. -4. **Synthesiser** streams an LLM completion using the prompt described in `prompt_inventory.md` (`retrieval_pipeline.synth_final`). - -## Configuration Flags (passed from UI โ†’ backend) -| Flag | Default | Effect | -|------|---------|--------| -| `searchType` | `fts` | UI label (FTS / Dense / Hybrid). | -| `retrievalK` | 10 | Initial candidate count per retriever. | -| `contextWindowSize` | 5 | How many adjacent chunks to merge (late-chunk). | -| `rerankerTopK` | 20 | How many docs to pass into AI reranker. | -| `denseWeight` | 0.5 | When `hybrid`, linear mix weight. | -| `aiRerank` | bool | Toggle reranker. | -| `verify` | bool | If true, pass answer to **Verifier** component. | +## Stage detail -## Interfaces -* Reads from **LanceDB** tables `text_pages_`. -* Calls **Ollama** generation model specified in `PIPELINE_CONFIGS`. -* Exposes `RetrievalPipeline.answer_stream()` iterator consumed by SSE API. +### 1. Retrieval โ€” `MultiVectorRetriever.retrieve()` (`retrievers.py`) -## Extension Points -* Plug new retriever by inheriting `BaseRetriever` and registering in `retrievers.py`. -* Swap reranker model via `EXTERNAL_MODELS['reranker_model']`. -* Custom answer prompt can be overridden by passing `prompt_override` to `_synthesize_final_answer()` (not yet surfaced in UI). +```python +retrieve(text_query: str, table_name: str, k: int, search_type: str = "hybrid", + *, where: Optional[str] = None) -> List[Dict] +``` -## Detailed Implementation Analysis +`where` is a LanceDB SQL predicate applied as a prefilter to every leg (metadata +filters, roadmap 4.4 โ€” see "Feature surface added 2026-08-14+" below). -### Core Architecture Pattern -The `RetrievalPipeline` uses **lazy initialization** for all components to avoid heavy memory usage during startup. Each component (embedder, retrievers, rerankers) is only loaded when first accessed via private `_get_*()` methods. +`search_type` selects which LanceDB legs run. Unknown values log a warning and degrade to `hybrid` (`retrievers.py:99-102`). -```python -def _get_text_embedder(self): - if self.text_embedder is None: - self.text_embedder = select_embedder( - self.config.get("embedding_model_name", "Qwen/Qwen3-Embedding-0.6B"), - self.ollama_config.get("host") - ) - return self.text_embedder -``` +| Mode | Legs | Score returned | +|------|------|----------------| +| `hybrid` (default) | FTS and vector, run concurrently on a 2-worker thread pool (`retrievers.py:139-143`), fused by reciprocal rank fusion | RRF score | +| `vector_only` | vector only | `1 / (1 + distance)` | +| `fts_only` | FTS only | LanceDB's BM25 score | -### Thread Safety Implementation -**Critical Issue**: ColBERT reranker and model loading are not thread-safe. The system uses multiple locks: +Details: -```python -# Global locks to prevent race conditions -_rerank_lock: Lock = Lock() # Protects .rank() calls -_ai_reranker_init_lock: Lock = Lock() # Prevents concurrent model loading -_sentence_pruner_lock: Lock = Lock() # Serializes Provence model init -``` +* **FTS leg** โ€” `tbl.search(query=..., query_type="fts").limit(k)`. Single-word queries are rewritten to `"* OR ~"` to add prefix and fuzzy matching (`retrievers.py:117-118`). This is LanceDB's built-in full-text index (created at index time, see `indexing_pipeline.md`) โ€” no SQLite, no Porter stemming, no configurable stop-word or n-gram handling. +* **Vector leg** โ€” the query is embedded once and memoised in a 256-entry `lru_cache` per retriever instance, then `tbl.search(vector).limit(k)`. Before searching, the retriever reads the table's embedder marker: a table written by a different embedding model raises `EmbedderMismatchError` (which is deliberately *not* swallowed by the catch-all below), and the query vector is L2-normalized only when the table's own vectors are, so LanceDB's default L2 ordering matches the cosine ordering the model cards specify. A table with no marker is searched the legacy, unnormalized way with a one-time warning. +* **Fusion** โ€” each leg contributes `1 / (60 + rank)` (`_RRF_K = 60`, `retrievers.py:20`). Rows are deduplicated on `chunk_id`, falling back to `_rowid` then `text` (`retrievers.py:83-89`), summed, sorted, and truncated to `k`. There is no weighted linear blend and no `dense_weight` knob. +* Each leg fetches `k` rows, so hybrid examines up to `2k` candidates and returns `k`. +* Every returned doc carries a finite, higher-is-better `score`. The raw per-leg values `bm25` and `_distance` are attached **only** when that leg actually hit the row. +* `metadata` is accepted as either a dict or a JSON string; `text` falls back through `metadata.original_text` โ†’ `row.text` โ†’ `""` (`retrievers.py:179-202`). +* On any exception other than `EmbedderMismatchError` the method logs and returns `[]`, so a missing table degrades to zero results rather than an error โ€” **unless** a `where` filter is set: a filtered search that fails re-raises, because "0 results" and "the restriction did not run" are different answers and only one of them is safe. An embedder mismatch propagates instead โ€” a wrong-model answer is worse than an error. -When multiple queries run in parallel, only one thread can initialize heavy models or perform reranking operations. +### 2. Late chunking at query time (`retrieval_pipeline.py:289-341`) -### Retrieval Strategy Deep-Dive +When the merged late-chunk config is enabled, two extra things happen: -#### 1. Multi-Vector Dense Retrieval (`_get_dense_retriever()`) -```python -self.dense_retriever = MultiVectorRetriever( - db_manager, # LanceDB connection - text_embedder, # Qwen3-Embedding embedder - vision_model=None, # Optional multimodal - fusion_config={} # Score combination rules -) -``` +1. A second `retrieve()` runs against the late-chunk table and its hits are appended to the candidate list (`:293-305`). The table name is `latechunk.lancedb_table_name` if set, otherwise `
` with `table_suffix` defaulting to `_lc` (`:85-92`) โ€” the same name `IndexingPipeline` writes. +2. **Sibling merging** (`:316-341`): for every retrieved chunk, the ยฑ1 neighbouring chunks from the same document are fetched from the main table and their text is concatenated into that chunk's `text`. This runs off the config flag alone, whether or not the `_lc` table exists. + +The config block is merged across both container names and both spellings โ€” `retrieval.late_chunking`, `retrieval.latechunk`, `retrievers.late_chunking`, `retrievers.latechunk`, later writes winning (`:70-83`) โ€” so a profile setting and a runtime API override can come from different places and both apply. -**Process**: -1. Query โ†’ embedding vector (1024D for Qwen3-Embedding-0.6B) -2. LanceDB ANN search using IVF-PQ index -3. Cosine similarity scoring -4. Returns top-K with metadata +The `default` profile enables late chunking at query time (`main.py:57-59`), while the RAG API's `/index` endpoint defaults `enable_latechunk` to `false` (`api_server.py:410`). Unless you explicitly index with late chunking on, the `_lc` table does not exist: the extra `retrieve()` logs "Could not search table โ€ฆ" and returns nothing, while sibling merging still applies. + +### 2b. Evidence-sufficiency retry (`RetrievalPipeline.retrieve_candidates`) + +_Roadmap item 2.1, shipped 2026-08-09. On in the `default` profile, off in `fast`._ + +One conditional second retrieval, and only one โ€” the evidence for iterative +retrieval says iteration 1โ†’2 captures ~95% of the gains and iterations โ‰ฅ3 are +noise. It wraps the first stage **and** the rerank stage, so it sees the ranking +the caller would actually have got. -#### 2. BM25 Full-Text Search (`_get_bm25_retriever()`) ```python -# Uses SQLite FTS5 under the hood -SELECT chunk_id, text, bm25(fts_table) as score -FROM fts_table -WHERE fts_table MATCH ? -ORDER BY bm25(fts_table) -LIMIT ? +"retrieval": {"retry": {"enabled": True, "min_top_score": 0.12, "max_attempts": 1}} ``` -**Token Processing**: -- Stemming via Porter algorithm -- Stop-word removal -- N-gram tokenization (configurable) +**The signal.** Not the raw top similarity. That was measured on the gold set and +is *anti*-correlated with success: the three `mixed` first-stage misses each +scored a **higher** top cosine than the median successful query, because absolute +similarity mostly encodes how close a query's phrasing sits to the corpus's +register. What carries signal is contrast โ€” how far the best candidate stands +above everything else the query dragged in: -#### 3. Hybrid Score Fusion -When both retrievers are enabled: -```python -final_score = (1 - dense_weight) * bm25_score + dense_weight * dense_score ``` -Default `dense_weight = 0.7` favors semantic over lexical matching (updated from 0.5). +evidence = (cos_top โˆ’ cos_background) / (1 โˆ’ cos_background) +``` -### Late-Chunk Merging Algorithm +`cos_background` is the mean cosine of candidates from rank 6 down; the +denominator rescales against this query's reachable headroom, keeping the result +in 0โ€“1 and comparable across queries. Cosine comes from LanceDB's squared-L2 +`_distance` on the L2-normalized v4 tables (`cos = 1 โˆ’ d/2`), so the retry is only +armed on a normalized table โ€” on a legacy table, or in `fts_only` mode, the score +is `None` and nothing fires. **RRF scores are never used**: they encode rank, not +confidence, and are near-identical for every query. -**Problem**: Small chunks lose context; large chunks dilute relevance. -**Solution**: Retrieve small chunks, then expand with neighbors. +When reranking is on, the top reranker score is preferred instead, but only when +it is a genuine 0โ€“1 probability (`QwenRerankerScorer` returns P("yes")); an +arbitrary logit is rejected rather than compared against a probability threshold. +The threshold for that path is `min_rerank_score`, defaulting to `min_top_score`. -```python -def _get_surrounding_chunks_lancedb(self, chunk, window_size): - start_index = max(0, chunk_index - window_size) - end_index = chunk_index + window_size - - sql_filter = f"document_id = '{document_id}' AND chunk_index >= {start_index} AND chunk_index <= {end_index}" - results = tbl.search().where(sql_filter).to_list() - - # Sort by chunk_index to maintain document order - return sorted(results, key=lambda x: x.get("chunk_index", 0)) -``` +**What happens on a fire.** `_reformulate_query()` makes one `format="json"` call +on the **enrichment** model asking for a rewrite in the vocabulary a document +would use, the whole first stage + rerank runs again on it, and **the better of +the two result sets by the same score is kept** โ€” a retry that does not improve +the evidence is discarded, never merged. A `retrieval_retry` event goes out +through `event_callback`, so the RAG API's SSE stream carries it and the UI +cascade shows a "Rechecking weak evidence" step. -**Benefits**: -- Maintains granular search precision -- Provides richer context for answer generation -- Configurable window size (default: 5 chunks = ~2500 tokens) +**Measured** (`../eval/decisions/phase2-pipeline.md`): fires on **9.7% of `mixed` +queries** (7/72) and 20.8% of `docs`, and moved `mixed` first-stage nDCG@10 from +0.889 to 0.901/0.906 across two runs with **zero per-query regressions**. It +repaired `docs_d16`, a genuine recall@10 miss. -### AI Reranker Implementation +### 3. AI reranking (`retrieval_pipeline.py`, `_rerank_stage`) -#### ColBERT Strategy (via rerankers-lib) -```python -from rerankers import Reranker -self.ai_reranker = Reranker("answerdotai/answerai-colbert-small-v1", model_type="colbert") +Loaded lazily behind `_ai_reranker_init_lock` so only one thread performs the heavy `from_pretrained()`; the `.rank()` call itself is serialised behind `_rerank_lock` because the `rerankers` backends are not thread-safe. -# Usage -scores = reranker.rank(query, [doc.text for doc in candidates]) -``` +Config keys read from `reranker`: -**ColBERT Architecture**: -- **Query encoding**: Each token โ†’ 128D vector -- **Document encoding**: Each token โ†’ 128D vector -- **Interaction**: MaxSim between all query-doc token pairs -- **Advantage**: Fine-grained token-level matching +| Key | Default in code | `default` profile | +|-----|-----------------|-------------------| +| `enabled` | falsy | **`true`** โ€” reranking is on by default since arm G, 2026-08-14 ([`../eval/DECISIONS.md`](../eval/DECISIONS.md)) | +| `model_name` | none โ€” missing logs a warning and skips reranking | `EXTERNAL_MODELS["reranker_model"]` = `Qwen/Qwen3-Reranker-4B`, loaded lazily on the first reranked query | +| `strategy` | `rerankers-lib` | `rerankers-lib` | +| `model_type` | `cross-encoder` | not set โ‡’ `cross-encoder` | +| `top_k` | all retrieved docs | `10` | +| `min_score` | unset (no threshold) | `0.5` โ€” Qwen scorer only: candidates the calibrated scorer marks below this P(relevant) against every query are dropped | +| `min_keep` | unset | `3` โ€” floor on candidates kept regardless of `min_score` | +| `top_percent` | unset | unset โ€” when set (0 < p โ‰ค 1) it overrides `top_k` with `max(1, len(docs) * p)` | -#### Fallback: BGE Cross-Encoder -```python -# When ColBERT fails/unavailable -from sentence_transformers import CrossEncoder -model = CrossEncoder('BAAI/bge-reranker-base') -scores = model.predict([(query, doc.text) for doc in candidates]) -``` +A model whose `model_type` is `qwen3`, or whose name contains `qwen3-reranker`, is routed to the in-repo `QwenRerankerScorer` โ€” the `rerankers` library builds Qwen3-Reranker with a randomly initialised score head, so this route is what makes the default reranker model correct rather than merely loadable. Otherwise `strategy: "rerankers-lib"` loads `rerankers.Reranker(model_name, model_type=model_type)`, and any other value constructs the local `CrossEncoderReranker` (`rerankers/reranker.py:5`), an `AutoModelForSequenceClassification` cross-encoder with batched scoring and an early-exit heuristic. -### Answer Synthesis Pipeline +**If the reranker fails to load, the pipeline logs `โš ๏ธ Could not load reranker '' (). Continuing without reranking.` and continues with the unranked candidates.** There is no second fallback model. -#### Prompt Engineering Pattern -```python -def _synthesize_final_answer(self, query: str, facts: str, *, event_callback=None): - prompt = f""" -You are an AI assistant specialised in answering questions from retrieved context. - -Context you receive -โ€ข VERIFIED FACTS โ€“ text snippets retrieved from the user's documents. -โ€ข ORIGINAL QUESTION โ€“ the user's actual query. - -Instructions -1. Evaluate each snippet for relevance to the ORIGINAL QUESTION -2. Synthesise an answer **using only information from relevant snippets** -3. If snippets contradict, mention the contradiction explicitly -4. If insufficient information: "I could not find that information in the provided documents." -5. Provide thorough, well-structured answer with relevant numbers/names -6. Do **not** introduce external knowledge - -โ€“โ€“โ€“โ€“โ€“ Retrieved Snippets โ€“โ€“โ€“โ€“โ€“ -{facts} -โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“โ€“ - -ORIGINAL QUESTION: "{query}" -""" - - response = self.llm_client.complete_stream( - prompt=prompt, - model=self.ollama_config["generation_model"] # qwen3:8b - ) - - for chunk in response: - if event_callback: - event_callback({"type": "answer_chunk", "content": chunk}) - yield chunk -``` +#### Decomposition: pooled first stage by default, sub-query scoring at rerank -**Advanced Features**: -- **Source Attribution**: Automatic citation generation -- **Confidence Scoring**: Based on retrieval scores and snippet relevance -- **Answer Verification**: Optional grounding check via Verifier component +_Roadmap item 2.2, shipped 2026-08-09; pooled first stage added by arm H, 2026-08-15._ -### Query Processing and Transformation +Decomposing the **first stage** dilutes it semantically; the 2026 evidence +(MultiConIR/SSRB) puts the win at the reranking stage instead. So: -#### Query Decomposition -```python -class QueryDecomposer: - def decompose_query(self, query: str) -> List[str]: - """Break complex queries into simpler sub-queries.""" - decomposition_prompt = f""" - Break down this complex question into 2-4 simpler sub-questions that would help answer the original question. - - Original question: {query} - - Sub-questions: - 1. - 2. - 3. - 4. - """ - - response = self.llm_client.complete( - prompt=decomposition_prompt, - model=self.enrichment_model # qwen3:0.6b for speed - ) - - # Parse response into list of sub-queries - return self._parse_subqueries(response) -``` +* The shipped default (arm H, 2026-08-15) is the **pooled first stage** + (`pooled_first_stage: true`, `compose_from_sub_answers: false`): each + sub-query runs first-stage retrieval, the candidates are pooled and + de-duplicated, and there is ONE rerank pass and ONE synthesis over the union + context (`_pooled_first_stage`). +* When sub-queries are supplied *and* the reranker is on (the default since + arm G, 2026-08-14), every candidate is + scored against **every** sub-query and the per-sub-query scores are combined + with `query_decomposition.rerank_aggregate` โ€” `"mean"` (default) or `"max"`. + `mean` is the default because it measured better than `max` on both halves of + the A/B ([`../eval/decisions/phase2-pipeline.md`](../eval/decisions/phase2-pipeline.md) ยง3). +* When the reranker is off there is no rerank stage, so the sub-queries only + drive the pooled first-stage fan-out. -#### HyDE (Hypothetical Document Embeddings) -```python -class HyDEGenerator: - def generate_hypothetical_doc(self, query: str) -> str: - """Generate hypothetical document that would answer the query.""" - hyde_prompt = f""" - Generate a hypothetical document passage that would perfectly answer this question: - - Question: {query} - - Hypothetical passage: - """ - - response = self.llm_client.complete( - prompt=hyde_prompt, - model=self.enrichment_model - ) - - return response.strip() -``` +The compose path is still available behind its pre-existing flag, +`query_decomposition.compose_from_sub_answers` (the UI "compose sub-answers" +toggle): it needs a separate *answer* per sub-question +to compose from, which a single shared candidate set cannot produce, so it runs a +full `RetrievalPipeline.run()` per sub-query in parallel. Exactly what runs: -### Caching and Performance Optimization +| `query_decomposition` | reranker | First stage | Rerank scored against | +|---|---|---|---| +| off | either | once, full query | full query | +| on, `compose_from_sub_answers: true` (opt-in) | either | **once per sub-query**, in parallel | that sub-query | +| on, pooled (profile default) | on | **once per sub-query**, pooled + deduped | **all sub-queries, aggregated** | +| on, pooled (profile default) | off | once per sub-query, pooled + deduped | โ€” (no rerank stage) | +| on, one sub-query after decomposition | either | once, the resolved query | the resolved query | -#### Semantic Query Caching -```python -class RetrievalPipeline: - def __init__(self, config, ollama_client, ollama_config): - # TTL cache for embeddings and results - self.query_cache = TTLCache(maxsize=100, ttl=300) # 5 min TTL - self.embedding_cache = LRUCache(maxsize=500) - self.semantic_threshold = 0.98 # Similarity threshold for cache hits - - def get_cached_result(self, query: str, session_id: str = None) -> Optional[Dict]: - """Check for semantically similar cached queries.""" - query_embedding = self._get_text_embedder().create_embeddings([query])[0] - - for cached_query, cached_data in self.query_cache.items(): - cached_embedding = cached_data["embedding"] - similarity = cosine_similarity([query_embedding], [cached_embedding])[0][0] - - if similarity > self.semantic_threshold: - # Check session scope if configured - if self.cache_scope == "session" and cached_data.get("session_id") != session_id: - continue - - print(f"๐ŸŽฏ Cache hit: {similarity:.3f} similarity") - return cached_data["result"] - - return None -``` +### 4. Context expansion (`RetrievalPipeline._run_after_candidates`) -#### Batch Processing Optimizations -```python -def process_query_batch(self, queries: List[str]) -> List[Dict]: - """Process multiple queries efficiently.""" - # Batch embed all queries - query_embeddings = self._get_text_embedder().create_embeddings(queries) - - # Batch search - results = [] - for i, query in enumerate(queries): - embedding = query_embeddings[i] - - # Search with pre-computed embedding - dense_results = self._search_dense_with_embedding(embedding) - bm25_results = self._search_bm25(query) - - # Combine and rerank - combined = self._combine_results(dense_results, bm25_results) - reranked = self._rerank_batch([query], [combined])[0] - - results.append(reranked) - - return results -``` +When the effective window size is greater than 0, each surviving doc is expanded with its neighbours from the same document via a metadata-only LanceDB filter (`document_id = ... AND chunk_index BETWEEN ...`), run across a thread pool. The union is deduplicated on `chunk_id` and re-sorted by `rerank_score`, then `_distance`, then `score`, then document order. -### Advanced Search Features +Two properties worth knowing: -#### Conversational Context Integration -```python -def answer_with_history(self, query: str, conversation_history: List[Dict], **kwargs): - """Answer query with conversation context.""" - # Build conversational context - context_prompt = self._build_conversation_context(conversation_history) - - # Expand query with context - expanded_query = f"{context_prompt}\n\nCurrent question: {query}" - - # Process with expanded context - return self.answer_stream(expanded_query, **kwargs) - -def _build_conversation_context(self, history: List[Dict]) -> str: - """Build context from conversation history.""" - context_parts = [] - - for turn in history[-3:]: # Last 3 turns for context - if turn.get("role") == "user": - context_parts.append(f"Previous question: {turn['content']}") - elif turn.get("role") == "assistant": - # Extract key points from previous answers - context_parts.append(f"Previous context: {turn['content'][:200]}...") - - return "\n".join(context_parts) -``` +* The rerank-score selection is applied to the **seed set, before expansion**: only chunks carrying a `rerank_score` (plus cross-reference hop chunks, which are appended after reranking by design) seed the expansion, so the freshly added neighbours survive rerank filtering. (The old code filtered the expanded set afterwards, which deleted every neighbour whenever the reranker ran.) +* When the caller passed a metadata filter, the expansion queries run **inside** that filter โ€” it is captured in the submitting thread (the executor's workers do not inherit thread-locals), so expansion can never pull in chunks outside the restricted corpus. -#### Multi-Index Search -```python -def search_multiple_indexes(self, query: str, index_ids: List[str], **kwargs): - """Search across multiple document indexes.""" - all_results = [] - - for index_id in index_ids: - table_name = f"text_pages_{index_id}" - - try: - # Search individual index - index_results = self._search_single_index(query, table_name, **kwargs) - - # Add index metadata - for result in index_results: - result["source_index"] = index_id - - all_results.extend(index_results) - - except Exception as e: - print(f"โš ๏ธ Error searching index {index_id}: {e}") - continue - - # Global reranking across all indexes - if len(all_results) > kwargs.get("retrieval_k", 20): - all_results = self._rerank_global(query, all_results, **kwargs) - - return all_results -``` +### 5. Provence sentence pruning (`retrieval_pipeline.py:448-462`) -### Error Handling and Resilience +Runs between context expansion and synthesis when `provence.enabled` is set. Loads `naver/provence-reranker-debertav3-v1` once behind `_sentence_pruner_lock`, drops sentences scoring below `provence.threshold` (default `0.1`, `:455`), and then removes any chunk whose text was pruned to nothing (`:460`). If the model cannot be downloaded or loaded, `SentencePruner` logs and `prune_documents()` echoes its input unchanged. -#### Graceful Degradation -```python -def answer_stream(self, query: str, **kwargs): - """Main answer method with comprehensive error handling.""" - try: - # Try full pipeline - return self._answer_stream_full_pipeline(query, **kwargs) - - except Exception as e: - print(f"โš ๏ธ Full pipeline failed: {e}") - - try: - # Fallback: Dense-only search - kwargs["search_type"] = "dense" - kwargs["ai_rerank"] = False - return self._answer_stream_fallback(query, **kwargs) - - except Exception as e2: - print(f"โš ๏ธ Fallback failed: {e2}") - - # Last resort: Direct LLM answer - return self._direct_llm_answer(query) - -def _direct_llm_answer(self, query: str): - """Direct LLM answer as last resort.""" - prompt = f""" - The document retrieval system is temporarily unavailable. - Please provide a helpful response acknowledging this limitation. - - User question: {query} - - Response: - """ - - response = self.llm_client.complete_stream( - prompt=prompt, - model=self.ollama_config["generation_model"] - ) - - yield "โš ๏ธ Document search unavailable. Providing general response:\n\n" - - for chunk in response: - yield chunk -``` +No shipped profile contains a `provence` block, so pruning is off unless a request enables it. -#### Recovery Mechanisms -```python -def recover_from_embedding_failure(self, query: str, **kwargs): - """Recover when embedding model fails.""" - print("๐Ÿ”„ Attempting embedding model recovery...") - - # Try to reinitialize embedder - try: - self.text_embedder = None # Clear failed instance - embedder = self._get_text_embedder() # Reinitialize - - # Test with simple query - test_embedding = embedder.create_embeddings(["test"]) - - if test_embedding is not None: - print("โœ… Embedding model recovered") - return True - - except Exception as e: - print(f"โŒ Recovery failed: {e}") - - # Fallback to BM25-only search - kwargs["search_type"] = "bm25" - kwargs["ai_rerank"] = False - print("๐Ÿ”„ Falling back to keyword search only") - - return False -``` +### 6. Synthesis (`retrieval_pipeline.py:223-261, 502-514`) -### Performance Monitoring and Metrics +The surviving chunk texts are joined with blank lines and passed to `_synthesize_final_answer()`, which streams a completion on the **generation** model. Each token is pushed to `event_callback("token", {"text": ...})` โ€” synthesis is push-based, not an iterator. -#### Query Performance Tracking -```python -class PerformanceTracker: - def __init__(self): - self.metrics = { - "query_count": 0, - "avg_response_time": 0, - "cache_hit_rate": 0, - "error_rate": 0, - "embedding_time": 0, - "retrieval_time": 0, - "reranking_time": 0, - "synthesis_time": 0 - } - - @contextmanager - def track_query(self, query: str): - """Context manager for tracking query performance.""" - start_time = time.time() - - try: - yield - - # Success metrics - duration = time.time() - start_time - self.metrics["query_count"] += 1 - self.metrics["avg_response_time"] = ( - (self.metrics["avg_response_time"] * (self.metrics["query_count"] - 1) + duration) - / self.metrics["query_count"] - ) - - except Exception as e: - # Error metrics - self.metrics["error_rate"] = ( - self.metrics["error_rate"] * self.metrics["query_count"] + 1 - ) / (self.metrics["query_count"] + 1) - - raise e - - finally: - self.metrics["query_count"] += 1 -``` +Before serialisation, `vector` and `_distance` are removed from every doc and NaN/Inf floats are nulled (`:479-500`). The return value is: -#### Resource Usage Monitoring -```python -def monitor_memory_usage(self): - """Monitor memory usage of pipeline components.""" - import psutil - import gc - - process = psutil.Process() - memory_info = process.memory_info() - - print(f"Memory Usage: {memory_info.rss / 1024 / 1024:.1f} MB") - - # Component-specific monitoring - if hasattr(self, 'text_embedder') and self.text_embedder: - print(f"Embedder loaded: {type(self.text_embedder).__name__}") - - if hasattr(self, 'ai_reranker') and self.ai_reranker: - print(f"Reranker loaded: {type(self.ai_reranker).__name__}") - - # Suggest cleanup if memory usage is high - if memory_info.rss > 8 * 1024 * 1024 * 1024: # 8GB - print("โš ๏ธ High memory usage detected - consider cleanup") - gc.collect() +```jsonc +{ "answer": "...", "source_documents": [ /* chunk dicts */ ] } ``` ---- +with `{"answer": "I could not find an answer in the documents.", "source_documents": []}` when nothing survives (`:476-477`). -## Configuration Reference +Since commit a3f999a, each snippet is prefixed with a `[Source document: โ€ฆ]` line +naming the file it came from, and synthesis rule 7 tells the model to name that +source when the question asks where something is defined or attribution matters. +Sources are also returned as the `source_documents` array and rendered by the UI as a collapsible list. -### Default Pipeline Configuration -```python -RETRIEVAL_CONFIG = { - "retriever": "multivector", - "search_type": "hybrid", - "retrieval_k": 20, - "reranker_top_k": 10, - "dense_weight": 0.7, - "late_chunking": { - "enabled": True, - "window_size": 5 - }, - "ai_rerank": True, - "verify_answers": False, - "cache_enabled": True, - "cache_ttl": 300, - "semantic_cache_threshold": 0.98 -} -``` +### 7. Semantic cache (`agent/loop.py:130-154, 305-324, 587-594`) -### Model Configuration -```python -MODEL_CONFIG = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", - "generation_model": "qwen3:8b", - "enrichment_model": "qwen3:0.6b", - "reranker_model": "answerdotai/answerai-colbert-small-v1", - "fallback_reranker": "BAAI/bge-reranker-base" -} -``` +Owned by `Agent`, not by the pipeline: -### Performance Tuning -```python -PERFORMANCE_CONFIG = { - "batch_sizes": { - "embedding": 32, - "reranking": 16, - "synthesis": 1 - }, - "timeouts": { - "embedding": 30, - "retrieval": 60, - "reranking": 30, - "synthesis": 120 - }, - "memory_limits": { - "max_cache_size": 1000, - "max_results_per_query": 100, - "chunk_size_limit": 2048 - } -} -``` +* `TTLCache(maxsize=100, ttl=300)` (`loop.py:33`) keyed by raw query text, storing `{embedding, result, session_id}`. +* Looked up by cosine similarity against the freshly embedded query; a hit requires similarity โ‰ฅ `semantic_cache_threshold` (`0.98` in both profiles). +* `cache_scope` defaults to `"session"`: entries from a different `session_id` are skipped (`loop.py:141`). Set it to `"global"` to share cached answers across sessions โ€” note that this can return one session's document-derived answer to another. +* Skipped entirely on the `direct_answer` route. +* Per-query embeddings are additionally memoised by the retriever's own 256-entry `lru_cache`. + +## Configuration flags -## Extension Examples +Both camelCase and snake_case are accepted; the RAG API normalises them to snake_case once at parse time (`api_server.py:51-76`). + +| Wire field | RAG API default when absent | `default` profile | `fast` profile | Effect | +|------------|-----------------------------|-------------------|----------------|--------| +| `retrieval_mode` (alias `search_type`) | not set โ‡’ profile value | `hybrid` | `vector_only` | `hybrid` / `vector_only` / `fts_only`. Anything else is rejected with HTTP 400 (`api_server.py:161-168`). | +| `retrieval_k` | not sent โ‡’ profile value | `20` | `10` | Rows fetched per leg, and the size of the fused candidate list. (The `20` UI default comes from the frontend, not the RAG API.) | +| `reranker_top_k` | not sent โ‡’ profile value | `10` | reranker off | Docs kept after reranking. | +| `context_window_size` | not sent โ‡’ profile value | `0` | `0` | Neighbouring chunks merged around each hit. The frontend sends `1` by default (`session-chat.tsx`), so UI traffic effectively uses `1`; a client that omits the field gets the profile's `0`. | +| `ai_rerank` | not sent โ‡’ profile value | reranker **enabled** (arm G) | reranker disabled | Toggles `reranker.enabled`. | +| `query_decompose` | not sent โ‡’ profile value | `true` | `false` | Toggles `query_decomposition.enabled`. | +| `compose_sub_answers` | not sent โ‡’ profile value | `false` (pooled first stage, arm H) | โ€” | Compose one answer from sub-answers vs. the default pooled candidates with a single synthesis. | +| `filters` | not sent โ‡’ unfiltered | โ€” | โ€” | Metadata filter object (roadmap 4.4), compiled to a LanceDB `where` prefilter on both legs; a malformed filter is a 400. A present filter also skips triage at the agent. | +| `context_expand` | not sent | โ€” | โ€” | `false` forces `window_size_override=0` for this request. | +| `verify` | not sent โ‡’ profile value | `true` | `false` | See `verifier.md`. | +| `force_rag` | `false` | โ€” | โ€” | Skip triage, force the RAG path; all other toggles still apply. | +| `provence_prune` | not sent โ‡’ disabled | absent | absent | Enable Provence sentence pruning. | +| `provence_threshold` | not sent โ‡’ `0.1` | absent | absent | Pruning threshold. Not exposed in the UI. | +| `model` | not sent | โ€” | โ€” | Per-request generation model. Applied through a context manager that restores the previous value afterwards, and ignored with a warning when the id does not suit the active backend (`api_server.py:88-113`). | + +UI defaults (`src/components/ui/session-chat.tsx`): compose `false`, decompose `true`, aiRerank `true`, contextExpand `true`, stream `true`, verify `true`, forceDocs `false`, provencePrune `false`, retrievalK `20`, contextWindowSize `1`, rerankerTopK `10`, searchType `hybrid`. + +There is no `dense_weight` / `denseWeight` knob anywhere in the stack, and no `fusion` config block. + +## Entry points -### Custom Retriever Implementation ```python -class CustomRetriever(BaseRetriever): - def search(self, query: str, k: int = 10) -> List[Dict]: - """Implement custom search logic.""" - # Your custom retrieval implementation - pass - - def get_embeddings(self, texts: List[str]) -> np.ndarray: - """Generate embeddings for custom retrieval.""" - # Your custom embedding logic - pass +# rag_system/pipelines/retrieval_pipeline.py +RetrievalPipeline.run(query, table_name=None, window_size_override=None, event_callback=None, + sub_queries=None, *, filters=None) -> Dict + +# rag_system/agent/loop.py +Agent.run(query, table_name=None, session_id=None, compose_sub_answers=None, query_decompose=None, + ai_rerank=None, context_expand=None, verify=None, retrieval_k=None, context_window_size=None, + reranker_top_k=None, retrieval_mode=None, force_rag=False, event_callback=None, + *, filters=None) -> Dict ``` -### Custom Reranker Implementation +`Agent.run` is a synchronous wrapper around `_run_async`. Both `/chat` and `/chat/stream` go through `Agent.run` โ€” including the `force_rag` path (`api_server.py:354-390`), so no request shape bypasses the toggles. `RetrievalPipeline` has no iterator/`answer_stream` entry point. + +Build them with the factory, never by hand: + ```python -class CustomReranker(BaseReranker): - def rank(self, query: str, documents: List[Dict]) -> List[Dict]: - """Implement custom reranking logic.""" - # Your custom reranking implementation - pass +from rag_system.factory import get_agent +agent = get_agent("default") # deep copy of PIPELINE_CONFIGS["default"] +result = agent.run("What does the contract say about termination?") ``` -### Custom Query Transformer -```python -class CustomQueryTransformer: - def transform(self, query: str, context: Dict = None) -> str: - """Transform query based on context.""" - # Your custom query transformation logic - pass -``` \ No newline at end of file +## Streaming event protocol + +`POST /chat/stream` returns `text/event-stream`; every frame is `data: {"type": , "data": }\n\n` (`api_server.py:329-333`). + +| Event | Payload | Emitted by | +|-------|---------|-----------| +| `analyze` | `{query}` | `loop.py:250-251` | +| `direct_answer` | `{}` | `loop.py:328-329` | +| `decomposition` | `{sub_queries}` | `loop.py:378-379` | +| `retrieval_started` | `{mode}` (pipeline) or `{count}` (decomposition) | `retrieval_pipeline.py:276-277`, `loop.py:384-385` | +| `retrieval_done` | `{count}` | `retrieval_pipeline.py:307-308`, `loop.py:401, 485` | +| `filters_applied` | `{spec, where, candidates}` โ€” only when a metadata filter was sent | `RetrievalPipeline._retrieve_candidates_filtered` | +| `retrieval_retry` | the retry info dict โ€” only when the evidence-sufficiency retry fires | `RetrievalPipeline._post_candidates` | +| `crossref_hop` | `{targets, chunks_added, ...}` โ€” only when `retrieval.crossref_hop` is on and fires | `RetrievalPipeline._crossref_hop` | +| `document_escalation` | the escalation payload (`document_name`, `chunks_used`, `approx_tokens`, signal/score/threshold/budget) | `rag_system/agent/escalation.py` | +| `rerank_started` / `rerank_done` | `{count}` | `retrieval_pipeline.py:346-347, 394-395`, `loop.py:402-403, 418-419, 486` | +| `context_expand_started` / `context_expand_done` | `{count}` | `retrieval_pipeline.py:404-405, 438-439` | +| `prune_started` / `prune_done` | `{count}` | `retrieval_pipeline.py:453-454, 461-462` | +| `token` | `{text}` | synthesis and composition streams | +| `sub_query_token` | `{index, text, question}` | `loop.py:430` | +| `sub_query_result` | `{index, query, answer, source_documents}` | `loop.py:453-459` | +| `single_query_result` / `final_answer` | the result dict | `loop.py:397-398, 533-534` | +| `complete` | the final result dict | `api_server.py:338` | +| `error` | `{error}` | `api_server.py:344` | + +## Interfaces + +* Reads LanceDB tables at `storage.lancedb_uri` (`./lancedb`), table `text_pages_` (`backend/database.py:351`) or the profile's `storage.text_table_name` (`text_pages_v4`) when a session has no linked index. +* Calls Ollama at `OLLAMA_CONFIG["host"]` โ€” generation model for answers, utility model for routing/decomposition/verification. +* Embeddings come from `select_embedder()`: a model name containing `/` is loaded from HuggingFace in-process, anything else is treated as an Ollama tag. `_get_text_embedder()` **raises** when `embedding_model_name` is missing rather than substituting a default, because a wrong-dimensionality embedder silently returns nothing useful against an existing index (`retrieval_pipeline.py:103-116`). +* Vector search is a brute-force scan: nothing in `rag_system/` ever calls `create_index` for an ANN/IVF-PQ index. Only the full-text index is built. + +## Extension points + +* **New retriever** โ€” the contract is duck-typed, not an ABC. Provide an object with `retrieve(text_query: str, table_name: str, k: int, search_type: str = "hybrid") -> List[Dict]` returning dicts with at least `chunk_id`, `text`, `score`, `document_id`, `chunk_index`, `metadata`, and return it from `RetrievalPipeline._get_dense_retriever()` (`retrieval_pipeline.py:118-133`). There is no `BaseRetriever` and no registry. +* **New reranker** โ€” either point `reranker.model_name` / `reranker.model_type` at another `rerankers`-library model, or set `reranker.strategy` to something other than `rerankers-lib` and swap the class constructed in `_get_ai_reranker()` (`retrieval_pipeline.py:135-165`). There is no `BaseReranker`. +* **Answer prompt** โ€” the synthesis prompt is an inline f-string at `retrieval_pipeline.py:225-250`. `_synthesize_final_answer(query, facts, *, event_callback=None)` takes no prompt-override argument; edit the literal. + +## Feature surface added 2026-08-14+ + +* **Metadata filters / DSL** (roadmap 4.4) โ€” a `filters` JSON object on `/chat`, `/chat/stream` and the CLI (`--filters`) is compiled by `rag_system/retrieval/filters.py::compile_filters` to a LanceDB `where` prefilter applied to both search legs (`MultiVectorRetriever.retrieve(..., where=...)`). The pipeline opens it as a thread-local `RetrievalPipeline.filter_scope`, emits `filters_applied`, and a filtered search that fails re-raises instead of returning `[]`. +* **Cross-leg dedupe** (2026-08-14) โ€” vector/FTS candidates are de-duplicated on `(document_id, chunk_index)` in the candidate-building path (`RetrievalPipeline.retrieve_candidates`), so the same passage no longer occupies two slots. +* **FTS quote-stripping** โ€” double quotes are stripped before the LanceDB FTS leg so a decomposer-emitted phrase cannot trip the FTS parser, and BM25 scores are read from the `_score` column (`retrievers.py`). +* **Cross-reference hop** (roadmap 4.2, off by default) โ€” `retrieval.crossref_hop`: after reranking, chunks from documents the results cross-reference are appended (`RetrievalPipeline._crossref_hop`, index-time extraction in `rag_system/indexing/crossref.py`); hop chunks are exempt from the rerank-score seed filter by design. +* **Overview prefilter** (roadmap 4.3, off by default) โ€” `retrieval.overview_prefilter` with `mode: "boost"` (reorder candidates toward overview-matched documents) or `"restrict"` (hide them). +* **Full-document escalation** (roadmap 4.1, off by default) โ€” `retrieval.document_escalation`: when evidence is still weak after the retry, the top-ranked chunk's whole document is reassembled in `chunk_index` order and appended to the synthesis context, capped at `token_budget` (`rag_system/agent/escalation.py`); emits `document_escalation`. +* **Synthesis context budgeting** (2026-08-14) โ€” rank-ordered docs are packed into an explicit token budget, `retrieval.synthesis_context_tokens` default **12000**, with sibling-span overlap suppression (`RetrievalPipeline._budget_synthesis_context`). +* **Embedding instruction override** โ€” the `embedding_instruction` config key or the `EMBEDDING_INSTRUCTION` env var overrides the query-side instruction prefix; set it to `""` to switch the prefix off (`RetrievalPipeline._query_instruction`). + +## Operational notes + +* The RAG API is a single-threaded `socketserver.TCPServer` (`api_server.py:527-530`), so requests are handled one at a time. Treat it as a single-concurrent-user service. +* `RAG_AGENT` and its `RetrievalPipeline` are process-wide singletons created once at startup (`api_server.py:34-37`). Per-request overrides write into that shared config object, but the agent snapshots the config before applying them and restores it afterwards, so an override is scoped to the request that sent it. +* Changing the embedding model requires re-indexing. `VectorIndexer` raises a clear error if you try to append vectors of a different width to an existing table (`indexing/embedders.py:110-116`), and the query side will simply fail to match if the dimensions differ. + +--- +_Keep this document updated when stages, config keys, or the event protocol change._ diff --git a/Documentation/system_overview.md b/Documentation/system_overview.md index 7c6aeac0..cc7d5c0c 100644 --- a/Documentation/system_overview.md +++ b/Documentation/system_overview.md @@ -1,429 +1,498 @@ -# ๐Ÿ—๏ธ RAG System - Complete System Overview +# ๐Ÿ—๏ธ localGPT โ€” Complete System Overview -_Last updated: 2025-01-09_ +_Last updated: 2026-08-08_ -This document provides a comprehensive overview of the Advanced Retrieval-Augmented Generation (RAG) System, covering its architecture, components, data flow, and operational characteristics. +A comprehensive overview of the localGPT Retrieval-Augmented Generation system: architecture, components, data flow, configuration and operational characteristics. Everything here was verified against the source in this repository; where a feature exists but is not wired up, it says so. --- ## 1. System Architecture -### 1.1 High-Level Architecture +### 1.1 High-level architecture -The RAG system implements a sophisticated 4-tier microservices architecture: +Four processes. The browser talks to **two** of them. ```mermaid graph TB subgraph "Client Layer" - Browser[๐Ÿ‘ค User Browser] - UI[Next.js Frontend
React/TypeScript] + Browser["๐Ÿ‘ค User Browser"] + UI["Next.js Frontend
React / TypeScript
Port 3000"] Browser --> UI end - - subgraph "API Gateway Layer" - Backend[Backend Server
Python HTTP Server
Port 8000] - UI -->|REST API| Backend + + subgraph "Gateway Layer" + Backend["Backend Server
backend/server.py
Port 8000"] end - + subgraph "Processing Layer" - RAG[RAG API Server
Document Processing
Port 8001] - Backend -->|Internal API| RAG + RAG["RAG API Server
rag_system/api_server.py
Port 8001"] end - + subgraph "LLM Service Layer" - Ollama[Ollama Server
LLM Inference
Port 11434] - RAG -->|Model Calls| Ollama + Ollama["Ollama
Port 11434"] end - + subgraph "Storage Layer" - SQLite[(SQLite Database
Sessions & Metadata)] - LanceDB[(LanceDB
Vector Embeddings)] - FileSystem[File System
Documents & Indexes] - - Backend --> SQLite - RAG --> LanceDB - RAG --> FileSystem + SQLite[("SQLite
sessions, messages, index metadata")] + LanceDB[("LanceDB
chunk vectors + native FTS")] + FileSystem["File system
shared_uploads/ ยท index_store/"] end + + UI -->|"REST"| Backend + UI -->|"SSE POST /chat/stream (default chat path)"| RAG + Backend -->|"POST /chat ยท POST /index"| RAG + Backend -->|"routing + direct answers"| Ollama + RAG -->|"generation ยท enrichment ยท verification"| Ollama + Backend --> SQLite + RAG -->|"index metadata only"| SQLite + RAG --> LanceDB + RAG --> FileSystem ``` -### 1.2 Component Breakdown +### 1.2 Component breakdown | Component | Technology | Port | Purpose | |-----------|------------|------|---------| -| **Frontend** | Next.js 15, React 19, TypeScript | 3000 | User interface, chat interactions | -| **Backend** | Python 3.11, HTTP Server | 8000 | API gateway, session management, routing | -| **RAG API** | Python 3.11, Advanced NLP | 8001 | Document processing, retrieval, generation | -| **Ollama** | Go-based LLM server | 11434 | Local LLM inference (embedding, generation) | -| **SQLite** | Embedded database | - | Sessions, messages, index metadata | -| **LanceDB** | Vector database | - | Document embeddings, similarity search | +| **Frontend** | Next.js 15, React 19, TypeScript, Tailwind v4 | 3000 | Chat UI, index management, retrieval settings | +| **Backend gateway** | Python 3.10+, `http.server` on `ThreadingTCPServer` | 8000 | Sessions, messages, uploads, index CRUD, first-layer routing | +| **RAG API** | Python 3.10+, `http.server` on `TCPServer` (serialized) | 8001 | Agent, retrieval pipeline, indexing pipeline | +| **Ollama** | External LLM server | 11434 | Generation and enrichment model inference | +| **SQLite** | Embedded | โ€“ | Sessions, messages, documents, index metadata | +| **LanceDB** | Embedded vector store | โ€“ | Chunk vectors + native full-text (BM25) index | + +For the process topology, request sequences and threading model see [`architecture_overview.md`](architecture_overview.md). --- ## 2. Core Functionality -### 2.1 Intelligent Dual-Layer Routing +### 2.1 Two-layer routing -The system's key innovation is its **dual-layer routing architecture** that optimizes both speed and intelligence: +Routing happens twice, in two different processes. Layer 1 is a deterministic gate with no model call; layer 2 is the system's single LLM routing layer. -#### **Layer 1: Speed Optimization Routing** -- **Location**: `backend/server.py` -- **Purpose**: Route simple queries to Direct LLM (~1.3s) vs complex queries to RAG Pipeline (~20s) -- **Decision Logic**: Pattern matching, keyword detection, query complexity analysis +#### Layer 1 โ€” gateway routing (`backend/server.py`, non-streaming path only) -```python -# Example routing decisions -"Hello!" โ†’ Direct LLM (greeting pattern) -"What does the document say about pricing?" โ†’ RAG Pipeline (document keyword) -"What's 2+2?" โ†’ Direct LLM (simple + short) -"Summarize the key findings from the report" โ†’ RAG Pipeline (complex + indicators) -``` +`should_use_rag(message, idx_ids, force_rag)` โ€” a module-level function, no LLM call, no network I/O: + +1. `force_rag` โ†’ RAG, unconditionally. +2. Session has **no linked indexes** โ†’ direct LLM, no RAG. +3. Message is unmistakable smalltalk (`hello`, `thanks!`, `bye`, `ok` โ€” a whole-message allowlist regex capped at six words) or a question about the assistant itself (`who are you`, `what model are you`) โ†’ direct LLM. +4. Everything else โ†’ RAG. + +This is retrieval-first: escalate rather than pre-decide. The gate deliberately over-sends to RAG because layer 2 can still answer directly โ€” the cost of a wrong "send to RAG" is one agent triage call, while the cost of a wrong "answer directly" is an unanswerable question. Pre-retrieval LLM routing was removed here in Phase 2.3: it is the weakest measured routing pattern (`Documentation/research/`), and it duplicated layer 2. The old `_simple_pattern_routing` keyword/length fallback (which misrouted any question containing the word "test") is gone with it. + +A `force_rag: true` field on the request skips the gate and always calls the RAG API. This layer does **not** run on the streaming path, because the browser calls the RAG API directly. + +#### Layer 2 โ€” agent triage (`rag_system/agent/loop.py`, always) + +`_triage_query_async()`: + +1. `_route_via_overviews()` โ€” the enrichment model classifies the query against the overviews loaded for the session, returning `direct_answer` or `rag_query`. +2. If conversation history exists for the session, short-circuit to `rag_query`. +3. Otherwise a fallback triage prompt (also on the enrichment model) picks `rag_query` or `direct_answer`. + +`force_rag=true` on the RAG API pins `query_type = "rag_query"` and skips all three steps, while still honouring the reranking / decomposition / verification toggles. + +Triage is two-way since the graph module was removed on 2026-08-09 (roadmap 2.5). `Agent._normalize_triage()` collapses anything that is not an explicit `direct_answer` โ€” including a stray `graph_query` from a small utility model โ€” to `rag_query`. + +### 2.2 Indexing + +1. **Upload** โ€” files are stored under `shared_uploads/` with a UUID prefix and recorded in SQLite. +2. **Conversion** โ€” `DocumentConverter` (Docling) produces markdown plus structure. OCR options are chosen by probing which backend is actually installed (OcrMac on macOS, then EasyOCR, RapidOCR, tesserocr, the `tesseract` CLI); when none is available Docling's defaults are used and text-layer PDFs still convert. +3. **Chunking** โ€” `DoclingChunker` packs sentences up to a token budget (`chunk_size`, default 512 over HTTP) using the embedding model's tokenizer. A legacy `MarkdownRecursiveChunker` exists and is selected by `chunker_mode: "legacy"` (used by `create_index_script.py`; over HTTP, `enable_docling_chunk: false` on `POST :8001/index` selects it โ€” the HTTP default is `true`). +4. **Overviews** โ€” the first *n* chunks of each document (default 5) are summarised by the enrichment model into `index_store/overviews/.jsonl`. The agent's triage router reads these; the gateway gate (ยง2.1) does not. +5. **Contextual enrichment** (optional) โ€” the enrichment model summarises a window of surrounding chunks and prepends it to each chunk. The untouched text is kept in `metadata.original_text`. +6. **Embedding + indexing** โ€” chunks are embedded and written to a LanceDB table; a native FTS index is created on the `text` column. +7. **Late chunking** (optional) โ€” each document is re-encoded in one pass and per-chunk vectors are pooled from it, written to `
_lc`. Enrichment produces *copies* of the chunks, so this leg encodes the original text, not the enriched text. + +The vector width is taken from the embeddings actually produced. If you point an existing table at a different-dimensional model, `VectorIndexer` raises with a message telling you to rebuild โ€” it will not silently corrupt the table. Because two models can share a width, each table also records the embedding model that wrote it and whether its vectors are L2-normalized; a mismatch raises `EmbedderMismatchError` at index time and at query time, and a table with no marker (built before this existed) keeps working unnormalized with a warning. + +### 2.3 Retrieval + +1. **Query embedding** โ€” the same embedding model used at index time (an LRU cache holds 256 single-query embeddings). Instruction-tuned families (harrier-oss-v1, Qwen3-Embedding) get the query-side `Instruct: โ€ฆ \nQuery: โ€ฆ` prefix their cards require; documents never do. The query vector is L2-normalized when the table's marker says its vectors are, so LanceDB's L2 ordering is the cosine ordering the cards specify. +2. **Search** โ€” `MultiVectorRetriever.retrieve()` runs LanceDB full-text search and vector search **in parallel** and fuses them with **reciprocal rank fusion** (`1/(60 + rank)` per leg). `search_type` selects `hybrid` (both legs), `vector_only` or `fts_only`; an unrecognised value logs a warning and falls back to `hybrid`. Every returned document carries a finite, higher-is-better `score`. +3. **Late-chunk leg** (optional) โ€” when late chunking is enabled the same query also runs against `
_lc` and those hits are appended. Every retrieved chunk (from either leg) then has its text replaced by the concatenation of its ยฑ1 neighbours in the main table, so a hit on one sub-vector still yields readable context. Hits are de-duplicated across the vector/FTS legs on `(document_id, chunk_index)` (shipped 2026-08-14). +4. **Reranking** (**on by default** since arm G, 2026-08-14 โ€” `reranker.enabled: True` with `min_score: 0.5` / `min_keep: 3` threshold-based selection; the measured reason is in [`../eval/DECISIONS.md`](../eval/DECISIONS.md)) โ€” the default `Qwen/Qwen3-Reranker-4B` goes to the in-repo `QwenRerankerScorer`, and any other model is loaded through the `rerankers` library (`reranker.strategy: "rerankers-lib"`, `reranker.model_type: "cross-encoder"`); any other `strategy` value uses the in-repo `CrossEncoderReranker` (`transformers` `AutoModelForSequenceClassification`). The model is loaded lazily on the first reranked query. If the model fails to load, a warning is printed and **reranking is skipped** โ€” there is no second reranker to fall back to. +5. **Context expansion** โ€” each surviving chunk is widened to its neighbours within `context_window_size`. +6. **Sentence pruning** (opt-in, `provence.enabled`) โ€” `naver/provence-reranker-debertav3-v1` drops sentences below `provence.threshold` (default `0.1`); chunks pruned to nothing are removed. +7. **Synthesis** โ€” the generation model writes the answer, streamed token by token. +8. **Verification** (see ยง2.4). + +Retrieval matches against the stored (possibly enriched) text, and chunks coming out of the retriever expose `metadata.original_text` as their `text` when enrichment ran โ€” so enrichment improves recall without pushing its own prefix into the answer's context. Neighbour chunks pulled in by context expansion are returned exactly as stored, so those still carry the enrichment prefix. + +### 2.4 Verification + +When `verification.enabled` is on (or `verify: true` is sent) **and** the result has source documents, `Verifier.verify_async()` asks the enrichment model for a JSON verdict and the confidence is appended to the answer **string**: + +* `" [Confidence: N%]"` for any non-zero score; +* additionally `" [Warning: Low confidence. Groundedness: ]"` when the answer is not grounded or the score is below 50; +* nothing at all when the score parses as 0 (treated as a parser failure). -#### **Layer 2: Intelligence Optimization Routing** -- **Location**: `rag_system/agent/loop.py` -- **Purpose**: Within RAG pipeline, route to optimal processing method -- **Methods**: - - `direct_answer`: General knowledge queries - - `rag_query`: Document-specific queries requiring retrieval - - `graph_query`: Entity relationship queries (future feature) - -### 2.2 Document Processing Pipeline - -#### **Indexing Process** -1. **Document Upload**: PDF files uploaded via web interface -2. **Text Extraction**: Docling library extracts text with layout preservation -3. **Chunking**: Intelligent chunking with configurable strategies (DocLing, Late Chunking, Standard) -4. **Embedding**: Text converted to vector embeddings using Qwen models -5. **Storage**: Vectors stored in LanceDB with metadata in SQLite - -#### **Retrieval Process** -1. **Query Processing**: User query analyzed and contextualized -2. **Embedding**: Query converted to vector embedding -3. **Search**: Hybrid search combining vector similarity and BM25 keyword matching -4. **Reranking**: AI-powered reranking for relevance optimization -5. **Synthesis**: LLM generates final answer using retrieved context - -### 2.3 Advanced Features - -#### **Query Decomposition** -- Complex queries automatically broken into sub-queries -- Parallel processing of sub-queries for efficiency -- Intelligent composition of final answers - -#### **Contextual Enrichment** -- Conversation history integration -- Context-aware query expansion -- Session-based memory management - -#### **Verification System** -- Answer verification against source documents -- Confidence scoring and grounding checks -- Source attribution and citation +There is no top-level `confidence` field in the response. Responses are `{answer, source_documents}`. + +### 2.5 Query decomposition + +Enabled in the `default` profile. The enrichment model splits the raw user query (plus up to 5 recent turns for pronoun resolution) into sub-queries, capped by `query_decomposition.max_sub_queries` (default 10). Sub-queries are retrieved in parallel with at most 3 worker threads. The shipped default (arm H, 2026-08-15) is the **pooled first stage** (`compose_from_sub_answers: false`, `pooled_first_stage: true`): the per-sub-query candidates are pooled and de-duplicated, then get ONE rerank pass and ONE synthesis over the union context. With `compose_from_sub_answers: true` the generation model instead answers each sub-query and composes a final answer from the sub-answers. A decomposition that yields a single sub-query skips the parallel machinery. + +### 2.6 Semantic cache and conversation memory + +* `TTLCache(maxsize=100, ttl=300)` keyed on the query embedding. A cached answer is reused when cosine similarity โ‰ฅ `semantic_cache_threshold` (`0.98`). +* `cache_scope` is `"session"` in both profiles: an entry is only reused inside the session that produced it. Setting it to `"global"` re-enables cross-session reuse โ€” including answers derived from another session's documents. +* Conversation history for triage and query rewriting lives in an in-process `LRUCache(maxsize=100)` on the agent, keyed by `session_id`. It is **not** the SQLite message history and is lost on restart. --- ## 3. Data Architecture -### 3.1 Storage Systems +### 3.1 SQLite (`backend/chat_data.db`, override with `DB_PATH`) -#### **SQLite Database** (`backend/chat_data.db`) ```sql --- Core tables -sessions -- Chat sessions with metadata -messages -- Individual messages and responses -indexes -- Document index metadata -session_indexes -- Links sessions to their indexes +sessions -- id, title, created_at, updated_at, model_used, message_count +messages -- id, session_id, content, sender('user'|'assistant'), timestamp, metadata +session_documents -- files uploaded to a session +indexes -- id, name, description, created_at, updated_at, vector_table_name, metadata +index_documents -- files belonging to a named index +session_indexes -- links sessions to indexes ``` -#### **LanceDB Vector Store** (`./lancedb/`) -``` -tables/ -โ”œโ”€โ”€ text_pages_[uuid] -- Document text embeddings -โ”œโ”€โ”€ image_pages_[uuid] -- Image embeddings (future) -โ””โ”€โ”€ metadata_[uuid] -- Document metadata -``` +Written by `backend/server.py`. The RAG API opens the same database but only reads/writes the `indexes` rows (index metadata); it never writes `messages`. + +### 3.2 LanceDB (`./lancedb`, override with `LANCEDB_PATH`) -#### **File System** (`./index_store/`) ``` -index_store/ -โ”œโ”€โ”€ overviews/ -- Document summaries for routing -โ”œโ”€โ”€ bm25/ -- BM25 keyword indexes -โ””โ”€โ”€ graph/ -- Knowledge graph data +lancedb/ +โ”œโ”€โ”€ text_pages_v4 -- default table (storage.text_table_name) +โ”œโ”€โ”€ text_pages_ -- one table per index created via POST /indexes +โ””โ”€โ”€
_lc -- late-chunk vectors for the table above ``` -### 3.2 Data Flow +Each table stores `chunk_id`, `text`, `document_id`, `chunk_index`, `metadata` (JSON) and `vector`, plus a native full-text index on `text`. -1. **Document Upload** โ†’ File System (`shared_uploads/`) -2. **Processing** โ†’ Embeddings stored in LanceDB -3. **Metadata** โ†’ Index info stored in SQLite -4. **Query** โ†’ Search LanceDB + SQLite coordination -5. **Response** โ†’ Message history stored in SQLite +### 3.3 File system ---- +``` +shared_uploads/ -- uploaded documents (_) +index_store/overviews/.jsonl -- per-index / per-session document overviews +index_store/overviews/overviews.jsonl -- global fallback overview file +logs/ -- run_system.py service logs + run_system.pid +``` -## 4. Model Architecture +--- -### 4.1 Configurable Model Pipeline +## 4. Models -The system supports multiple embedding and generation models with automatic switching: +### 4.1 Configured defaults (`rag_system/main.py`) -#### **Current Model Configuration** ```python -EXTERNAL_MODELS = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", # 1024D - "reranker_model": "answerdotai/answerai-colbert-small-v1", # ColBERT reranker - "vision_model": "Qwen/Qwen-VL-Chat", # Vision model for multimodal - "fallback_reranker": "BAAI/bge-reranker-base", # Backup reranker +OLLAMA_CONFIG = { + "host": os.getenv("OLLAMA_HOST", "http://localhost:11434"), + "generation_model": os.getenv("GENERATION_MODEL", "qwen3.5:9b"), + "enrichment_model": os.getenv("ENRICHMENT_MODEL", "qwen3.5:4b"), } -OLLAMA_CONFIG = { - "generation_model": "qwen3:8b", # High-quality generation - "enrichment_model": "qwen3:0.6b", # Fast enrichment/routing - "host": "http://localhost:11434" +EXTERNAL_MODELS = { + "embedding_model": os.getenv("EMBEDDING_MODEL", "microsoft/harrier-oss-v1-0.6b"), + "reranker_model": os.getenv("RERANKER_MODEL", "Qwen/Qwen3-Reranker-4B"), } ``` -#### **Model Switching** -- **Per-Session**: Each chat session can use different embedding models -- **Automatic**: System automatically switches models based on index metadata -- **Dynamic**: Models loaded just-in-time to optimize memory usage +| Role | Default | Used for | +|------|---------|----------| +| Generation | `qwen3.5:9b` (Ollama) | Final answers, sub-answer composition, direct answers | +| Enrichment / utility | `qwen3.5:4b` (Ollama) | Agent triage (the only LLM router), query decomposition, contextual enrichment, document overviews, verification | +| Embedding | `microsoft/harrier-oss-v1-0.6b` (HuggingFace, MIT, 1024 dims) | Index and query embeddings | +| Reranker | `Qwen/Qwen3-Reranker-4B` (HuggingFace, own yes/no-logit scorer) | Reranking retrieved chunks โ€” **on by default** (arm G, 2026-08-14), loaded lazily on the first reranked query ([`../eval/DECISIONS.md`](../eval/DECISIONS.md)) | +| Sentence pruner | `naver/provence-reranker-debertav3-v1` (HuggingFace) | Opt-in sentence-level pruning | + +Approximate footprints published by the model authors (not measured here): `qwen3.5:9b` โ‰ˆ 6.6 GB at Q4, `qwen3.5:4b` โ‰ˆ 3.4 GB, `qwen3.6:27b` โ‰ˆ 17 GB, `microsoft/harrier-oss-v1-0.6b` โ‰ˆ 1.2 GB, `Qwen/Qwen3-Embedding-4B` โ‰ˆ 8 GB in bf16, `Qwen/Qwen3-Embedding-0.6B` โ‰ˆ 1.2 GB, `Qwen/Qwen3-Reranker-4B` โ‰ˆ 7.5 GB. + +### 4.2 Documented alternatives + +| Role | Options | +|------|---------| +| Generation | `qwen3.6:27b` (high-end), `qwen3.5:4b` (light) | +| Enrichment | `qwen3.5:2b` (light) | +| Embedding | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context โ€” for multilingual / long-context corpora), `Qwen/Qwen3-Embedding-0.6B` (1024 dims, light) | +| Reranker | `BAAI/bge-reranker-v2-m3` (cross-encoder, low latency โ€” only pays off with a weaker embedder than the default), `answerdotai/answerai-colbert-small-v1` (late interaction โ€” also set `reranker.model_type: "colbert"`), `Qwen/Qwen3-Reranker-0.6B` | -### 4.2 Supported Models +Set them with the `GENERATION_MODEL` / `ENRICHMENT_MODEL` / `EMBEDDING_MODEL` / `RERANKER_MODEL` environment variables, or edit `rag_system/main.py`. -#### **Embedding Models** -- `Qwen/Qwen3-Embedding-0.6B` (1024D) - Default, fast and high-quality +> โš ๏ธ **Changing the embedding model requires re-indexing.** Vector width is derived from the loaded model, and `VectorIndexer` raises rather than appending mismatched vectors to an existing LanceDB table. Width alone is not a sufficient check โ€” `harrier-oss-v1-0.6b` and `Qwen3-Embedding-0.6B` are both 1024 dims โ€” so every table also records the embedding model that wrote it, and indexing into or querying it with a different one raises `EmbedderMismatchError`. Ollama embedding tags are also supported: `select_embedder()` treats a name containing `/` as a HuggingFace repo and anything else as an Ollama tag. -#### **Generation Models** (via Ollama) -- `qwen3:8b` - Primary generation model (high quality) -- `qwen3:0.6b` - Fast enrichment and routing model +### 4.3 Model selection at runtime -#### **Reranking Models** -- `answerdotai/answerai-colbert-small-v1` - Primary ColBERT reranker -- `BAAI/bge-reranker-base` - Fallback cross-encoder reranker +* **Per request** โ€” `model` on `POST :8000/sessions/{id}/messages` and on the RAG API chat endpoints overrides the generation model for that request only. The RAG API rejects ids that do not match the active backend (an Ollama tag will not be forced onto a WatsonX deployment). +* **Per index** โ€” when an index records an `embedding_model` in its metadata, the RAG API switches the retrieval pipeline's embedder to it before querying that index. +* Generation model precedence on the gateway's direct-LLM path: request `model` โ†’ the session's `model_used` โ†’ `GENERATION_MODEL`. -#### **Vision Models** (Multimodal) -- `Qwen/Qwen-VL-Chat` - Vision-language model for image processing +### 4.4 Vision / multimodal โ€” not integrated + +There is **no** vision model in the configuration and no multimodal path in the pipelines: PDF parsing and OCR are handled entirely by Docling. Models such as GLM-OCR or Qwen3-VL could be added as an extension; wiring them up is not done today. + +### 4.5 Alternative LLM backend: WatsonX + +`LLM_BACKEND=watsonx` swaps the Ollama client for `WatsonXClient` (`WATSONX_CONFIG`: `WATSONX_API_KEY`, `WATSONX_PROJECT_ID`, `WATSONX_URL`, `WATSONX_GENERATION_MODEL`, `WATSONX_ENRICHMENT_MODEL`). It requires `pip install ibm-watsonx-ai` โ€” the root `requirements.txt` lists it as an optional, commented dependency. Embedding and reranking still run locally through HuggingFace. See [`../WATSONX_README.md`](../WATSONX_README.md). --- ## 5. Pipeline Configurations -### 5.1 Default Production Pipeline +`PIPELINE_CONFIGS` in `rag_system/main.py` contains exactly two profiles, `default` and `fast`. `RAG_CONFIG_MODE` selects the one the RAG API server uses (default `default`); an unknown value silently falls back to `default`. `factory.get_pipeline_config()` hands out a deep copy, so runtime overrides never mutate the master config. + +### 5.1 `default` + +```python +"default": { + "description": "Production-ready pipeline with hybrid search, query decomposition, and verification", + "storage": { + "lancedb_uri": "./lancedb", + "text_table_name": "text_pages_v4" + }, + "retrieval": { + "search_type": "hybrid", + "latechunk": {"enabled": True}, + "dense": {"enabled": True}, + "retry": {"enabled": True, "min_top_score": 0.12, "max_attempts": 1}, + # Phase-4 features, all off until benchmarked: + "document_escalation": {"enabled": False, "max_documents": 1, "token_budget": 6000}, + "crossref_hop": {"enabled": False, "max_hops": 1, "chunks_per_hop": 3}, + "overview_prefilter": {"enabled": False, "top_documents": 5, "mode": "boost"} + }, + "embedding_model_name": EXTERNAL_MODELS["embedding_model"], + "reranker": { + "enabled": True, # arm G, 2026-08-14 โ€” see eval/DECISIONS.md + "model_type": "cross-encoder", + "strategy": "rerankers-lib", + "model_name": EXTERNAL_MODELS["reranker_model"], + "top_k": 10, + "min_score": 0.5, + "min_keep": 3 + }, + # Arm H (2026-08-15): pooled first stage is the shipped default โ€” per-sub-query + # retrieval, pooled + deduped candidates, ONE rerank + ONE synthesis. + "query_decomposition": { + "enabled": True, + "compose_from_sub_answers": False, + "pooled_first_stage": True + }, + "verification": {"enabled": True}, + "retrieval_k": 20, + "context_window_size": 0, + "semantic_cache_threshold": 0.98, + "cache_scope": "session", + "contextual_enricher": {"enabled": True, "window_size": 1}, + "indexing": { + "embedding_batch_size": 50, + "enrichment_batch_size": 10, + "extract_crossrefs": True + } +} +``` + +### 5.2 `fast` ```python -PIPELINE_CONFIGS = { - "default": { - "description": "Production-ready pipeline with hybrid search, AI reranking, and verification", - "storage": { - "lancedb_uri": "./lancedb", - "text_table_name": "text_pages_v3", - "bm25_path": "./index_store/bm25", - "graph_path": "./index_store/graph/knowledge_graph.gml" - }, - "retrieval": { - "retriever": "multivector", - "search_type": "hybrid", - "late_chunking": { - "enabled": True, - "table_suffix": "_lc_v3" - }, - "dense": { - "enabled": True, - "weight": 0.7 - }, - "bm25": { - "enabled": True, - "index_name": "rag_bm25_index" - } - }, - "embedding_model_name": "Qwen/Qwen3-Embedding-0.6B", - "reranker": { - "enabled": True, - "model_name": "answerdotai/answerai-colbert-small-v1", - "top_k": 20 - } +"fast": { + "description": "Speed-optimized pipeline with minimal overhead", + "storage": {"lancedb_uri": "./lancedb", "text_table_name": "text_pages_v4"}, + "retrieval": { + "search_type": "vector_only", + "latechunk": {"enabled": False}, + "dense": {"enabled": True} + }, + "embedding_model_name": EXTERNAL_MODELS["embedding_model"], + "reranker": {"enabled": False}, + "query_decomposition": {"enabled": False}, + "verification": {"enabled": False}, + "retrieval_k": 10, + "context_window_size": 0, + "semantic_cache_threshold": 0.98, + "cache_scope": "session", + "contextual_enricher": {"enabled": False, "window_size": 1}, + "indexing": { + "embedding_batch_size": 100, + "enrichment_batch_size": 50 } } ``` -### 5.2 Processing Options +One key in the blocks above currently has no consumer and is inert: the profile's `description`. (`indexing.enable_progress_tracking` used to be listed here; it was assigned to an attribute that is never checked โ€” progress is always tracked โ€” and has since been removed from the profiles.) -#### **Chunking Strategies** -- **Standard**: Fixed-size chunks with overlap -- **DocLing**: Structure-aware chunking using DocLing library -- **Late Chunking**: Small chunks expanded at query time +### 5.3 Keys read at runtime but absent from the profiles -#### **Enrichment Options** -- **Contextual Enrichment**: AI-generated chunk summaries -- **Overview Building**: Document-level summaries for routing -- **Graph Extraction**: Entity and relationship extraction +These have code defaults and can be added to a profile if you want to change them: ---- +| Key | Default | Effect | +|-----|---------|--------| +| `chunking.chunk_size` | `1500` (profile absent) / `512` (HTTP requests) | Token budget per chunk | +| `chunker_mode` | `"docling"` | `"docling"` or `"legacy"` | +| `query_decomposition.max_sub_queries` | `10` | Cap on sub-queries | +| `query_decomposition.rerank_aggregate` | `"mean"` | `mean` or `max`; how per-sub-query rerank scores combine (roadmap 2.2) | +| `retrieval.retry.min_rerank_score` | falls back to `min_top_score` | Retry threshold used when the reranker returns a 0โ€“1 probability | +| `verification.model` / `VERIFIER_MODEL` | unset | HuggingFace NLI/verifier model; unset keeps the LLM-prompt verifier (roadmap 2.4) | +| `verification.threshold` | `0.5` | Grounded/ungrounded cut for the local verifier | +| `reranker.model_type` | `"cross-encoder"` | `rerankers` library model type | +| `reranker.top_percent` | โ€“ | Keep a fraction of candidates instead of `top_k` | +| `provence.enabled` / `provence.threshold` | `False` / `0.1` | Sentence-level pruning | +| `overview.enabled` / `overview.model` / `overview.max_chunks` | `True` / enrichment model / `5` | Document overview generation | +| `enrich_model` | enrichment model | Overrides the model used for contextual enrichment | +| `overview_path` | `index_store/overviews/overviews.jsonl` | Where overviews are written | -## 6. Performance Characteristics +### 5.4 Where the 20/1/10 request defaults come from -### 6.1 Response Times +The `retrieval_k: 20`, `context_window_size: 1` and `reranker_top_k: 10` defaults are owned by the **frontend** (`src/components/ui/session-chat.tsx`), not by the RAG API. `rag_system/api_server.py` passes `None` for every option the client omits โ€” `verify`, `ai_rerank`, `query_decompose`, `compose_sub_answers`, `context_expand`, `retrieval_mode` and the three values above โ€” so the profile wins. -| Operation | Time Range | Notes | -|-----------|------------|-------| -| Simple Chat | 1-3 seconds | Direct LLM, no retrieval | -| Document Query | 5-15 seconds | Includes retrieval and reranking | -| Complex Analysis | 15-30 seconds | Multi-step reasoning | -| Document Indexing | 2-5 min/100MB | Depends on enrichment settings | +Note the practical consequence: for UI clients, context expansion of ยฑ1 chunk is on by default even though both profiles set `context_window_size: 0`. A non-UI HTTP client that omits the field gets the profile value (`0`). -### 6.2 Memory Usage +--- + +## 6. Resource Notes -| Component | Memory Usage | Notes | -|-----------|--------------|-------| -| Embedding Model | 1-2GB | Qwen3-Embedding-0.6B | -| Generation Model | 8-16GB | qwen3:8b | -| Reranker Model | 500MB-1GB | ColBERT reranker | -| Database Cache | 500MB-2GB | LanceDB and SQLite | +There are no benchmarks in this repository, so no latency or throughput figures are published here. What determines cost: -### 6.3 Scalability +* **Memory** is dominated by the models you load: the Ollama generation model, plus the embedding model and (if enabled) the reranker and Provence pruner, which run in the RAG API process via `transformers`. +* **Concurrency** is bounded by the RAG API's single-threaded server: one RAG request at a time per process. The backend gateway is threaded, so session and index CRUD stay responsive while a query runs. +* **Indexing cost** scales with contextual enrichment (one LLM call per chunk โ€” the contextualizer loops chunks inside each batch) and late chunking (a second full encode of every document, plus a second vector table). +* **Query cost** scales with query decomposition (one retrieval per sub-query, then one rerank and one synthesis over the pooled candidates), reranking and verification. The `fast` profile turns all of these off. -- **Concurrent Users**: 5-10 users with 16GB RAM -- **Document Capacity**: 10,000+ documents per index -- **Query Throughput**: 10-20 queries/minute per instance -- **Storage**: Approximately 1MB per 100 pages indexed +Use `python system_health_check.py` to print the resolved configuration, the embedding dimension of the loaded model, and the LanceDB tables that actually exist. --- -## 7. Security & Privacy +## 7. Configuration -### 7.1 Data Privacy +### 7.1 Environment variables -- **Local Processing**: All AI models run locally via Ollama -- **No External Calls**: No data sent to external APIs -- **Document Isolation**: Documents stored locally with session-based access -- **User Isolation**: Each session maintains separate context +Every variable below is read by this repository's code, except `HF_TOKEN` which is consumed by the HuggingFace client libraries. See [`.env.example`](../.env.example) for the annotated file. ---- +| Variable | Default | Read by | +|----------|---------|---------| +| `OLLAMA_HOST` | `http://localhost:11434` | `rag_system/main.py`, `backend/ollama_client.py` | +| `RAG_API_URL` | `http://localhost:8001` | `backend/server.py` (all calls to the RAG API) | +| `NEXT_PUBLIC_API_URL` | `http://localhost:8000` | `src/lib/api.ts` โ€” **inlined at build time** | +| `NEXT_PUBLIC_RAG_API_URL` | `http://localhost:8001` | `src/lib/api.ts` โ€” **inlined at build time** | +| `DB_PATH` | `backend/chat_data.db` (`/app/backend/chat_data.db` in Docker) | `backend/database.py` | +| `LANCEDB_PATH` | `storage.lancedb_uri`, else `./lancedb` | `rag_system/main.py` (pipeline profiles), `backend/database.py`, `system_health_check.py` | +| `GENERATION_MODEL` | `qwen3.5:9b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | same | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | `rag_system/main.py` | +| `RERANKER_MODEL` | `Qwen/Qwen3-Reranker-4B` (loaded lazily on the first reranked query) | `rag_system/main.py` | +| `RAG_CONFIG_MODE` | `default` | `rag_system/api_server.py` | +| `LLM_BACKEND` | `ollama` | `rag_system/main.py`, `rag_system/factory.py` | +| `RAG_API_TIMEOUT` | `600` (seconds) | `backend/server.py` โ€” chat calls | +| `RAG_API_INDEX_TIMEOUT` | `3600` (seconds) | `backend/server.py` โ€” indexing calls | +| `HF_TOKEN` | โ€“ | `huggingface_hub` (library) โ€” gated model downloads | -## 8. Configuration & Customization +`NEXT_PUBLIC_*` values are baked into the JavaScript bundle by `next build`; changing them at runtime has no effect on an already-built frontend. The compose files pass them as build args as well as runtime environment. -### 8.1 Model Configuration -Models can be configured in `rag_system/main.py`: +Service **ports** are not environment-configurable: `PORT = 8000` in `backend/server.py`, `8001` in `start_server()`, `3000` from Next.js. -```python -# Embedding model configuration -EXTERNAL_MODELS = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", # Your preferred model - "reranker_model": "answerdotai/answerai-colbert-small-v1", -} +### 7.2 Per-request options -# Generation model configuration -OLLAMA_CONFIG = { - "generation_model": "qwen3:8b", # Your LLM model - "enrichment_model": "qwen3:0.6b", # Your fast model -} -``` +Retrieval and indexing behaviour is controlled per request, not by editing a config file. See [`api_reference.md`](api_reference.md) for the full field list. Both casings are accepted end to end: the frontend historically sent camelCase, the gateway sends snake_case, and both the gateway and the RAG API normalise every option to one canonical snake_case key at parse time. -### 8.2 Pipeline Configuration -Processing behavior configured in `PIPELINE_CONFIGS`: +### 7.3 Command-line entry points -```python -PIPELINE_CONFIGS = { - "retrieval": { - "search_type": "hybrid", - "dense": {"weight": 0.7}, - "bm25": {"enabled": True} - }, - "chunking": { - "chunk_size": 512, - "chunk_overlap": 64, - "enable_latechunk": True, - "enable_docling": True - } -} +```bash +# Index a file or a directory (walks for .pdf .docx .html .htm .md .txt) +python -m rag_system.main index /path/to/docs --mode default + +# One-shot query, prints JSON +python -m rag_system.main chat "What does the contract say about termination?" --mode default + +# Start the RAG API +python -m rag_system.main api --port 8001 ``` -### 8.3 UI Configuration -Frontend behavior configured in environment variables: +`python rag_system/main.py โ€ฆ` does **not** work โ€” the module must be run with `-m` from the project root. Programmatically, `from rag_system.factory import get_agent, get_indexing_pipeline` is the supported entry point; `IndexingPipeline.run(file_paths)` is the public indexing call. + +--- + +## 8. Operations + +### 8.0 Prerequisites + +* **Python 3.10+** (3.11 recommended โ€” both Docker images are `python:3.11-slim`). +* **Node 20+** for the frontend (`Dockerfile.frontend` is `node:20-alpine`). +* **Ollama** installed and running, with the generation and enrichment models pulled. +* Python dependencies: `pip install -r requirements.txt`. `backend/requirements.txt` is the minimal set for running only the gateway. `pip install ibm-watsonx-ai` is additionally required for `LLM_BACKEND=watsonx`. + +### 8.1 Local launcher ```bash -NEXT_PUBLIC_API_URL=http://localhost:8000 -NEXT_PUBLIC_ENABLE_STREAMING=true -NEXT_PUBLIC_MAX_FILE_SIZE=50MB +python run_system.py # dev mode: all four services +python run_system.py --mode prod # runs `npm run build` before `npm run start` +python run_system.py --no-frontend # backend stack only +python run_system.py --health # HTTP probes each service, exits non-zero if unhealthy +python run_system.py --stop # terminates the processes recorded in logs/run_system.pid +python run_system.py --logs-only # tails logs/*.log without starting anything ``` ---- +`--health` probes `http://localhost:11434/api/tags`, `:8001/health`, `:8000/health` and `:3000/`. On startup the launcher checks that `GENERATION_MODEL` and `ENRICHMENT_MODEL` are present in Ollama. + +### 8.2 Docker -## 9. Monitoring & Observability +`docker compose --env-file docker.env up -d --build` brings up `rag-api`, `backend` and `frontend`; Ollama runs on the host by default (`OLLAMA_HOST=http://host.docker.internal:11434`, with `extra_hosts: host.docker.internal:host-gateway` so it also resolves on Linux). A containerised Ollama is available behind the `with-ollama` profile. `backend` and `rag-api` bind-mount `./backend`, `./lancedb`, `./index_store` and `./shared_uploads`, so both processes share one SQLite file and one vector store. Health checks use `/health` on both Python services and busybox `wget` for the frontend. See [`docker_usage.md`](docker_usage.md) and [`../DOCKER_README.md`](../DOCKER_README.md). -### 9.1 Logging System -- **Structured Logging**: JSON-formatted logs with timestamps -- **Log Levels**: DEBUG, INFO, WARNING, ERROR -- **Log Rotation**: Automatic log file rotation -- **Component Isolation**: Separate logs per service +### 8.3 Health and logging -### 9.2 Health Monitoring -- **Health Endpoints**: `/health` on all services -- **Service Dependencies**: Cascading health checks -- **Performance Metrics**: Response times, error rates -- **Resource Monitoring**: Memory, CPU, disk usage +| Endpoint | Response | +|----------|----------| +| `GET :8000/health` | `{status, ollama_running, available_models, database_stats}` | +| `GET :8001/health` | `{"status": "ok"}` | -### 9.3 Debugging Features -- **Debug Mode**: Detailed operation tracing -- **Query Inspection**: Step-by-step query processing -- **Model Switching Logs**: Embedding model change tracking -- **Error Reporting**: Comprehensive error context +`run_system.py` writes per-service logs to `logs/.log` plus `logs/system.log`, with a coloured console formatter. Logging is plain text โ€” there is no JSON formatter and no log rotation (see [`improvement_plan.md`](improvement_plan.md)). The RAG API routes its handler output through the `logging` module; the agent and both pipelines still print progress with `print()`, which is what you see in `logs/rag-api.log`. --- -## โš™๏ธ Configuration Modes - -The system supports multiple configuration modes optimized for different use cases: - -### **Default Mode** (`"default"`) -- **Description**: Production-ready pipeline with full features -- **Search**: Hybrid (dense + BM25) with 0.7 dense weight -- **Reranking**: AI-powered ColBERT reranker -- **Query Processing**: Query decomposition enabled -- **Verification**: Grounding verification enabled -- **Performance**: ~3-8 seconds per query -- **Memory**: ~10-16GB (with models loaded) - -### **Fast Mode** (`"fast"`) -- **Description**: Speed-optimized pipeline with minimal overhead -- **Search**: Vector-only (no BM25, no late chunking) -- **Reranking**: Disabled -- **Query Processing**: Single-pass, no decomposition -- **Verification**: Disabled -- **Performance**: ~1-3 seconds per query -- **Memory**: ~8-12GB (with models loaded) - -### **BM25 Mode** (`"bm25"`) -- **Description**: Traditional keyword-based search -- **Search**: BM25 only -- **Use Case**: Exact keyword matching, legacy compatibility - -### **Graph RAG Mode** (`"graph_rag"`) -- **Description**: Knowledge graph integration (currently disabled) -- **Status**: Available for future implementation -- **Use Case**: Relationship-aware retrieval +## 9. Security & Privacy + +* **Local by default** โ€” generation, embedding, reranking and pruning all run locally (Ollama + HuggingFace models). Nothing leaves the machine unless you set `LLM_BACKEND=watsonx`, which sends prompts to IBM Cloud. +* **Model downloads** โ€” HuggingFace models are fetched on first use and cached; that is the only outbound traffic in the default setup. +* **No authentication** โ€” neither server implements auth, and both send `Access-Control-Allow-Origin: *`. Ports 8000 and 8001 must not be exposed to an untrusted network. +* **Session isolation** โ€” retrieval is scoped to the tables of the indexes linked to a session, and the semantic cache is session-scoped by default. Setting `cache_scope: "global"` allows one session's document-derived answer to be returned in another. +* **Deletion** โ€” deleting an index removes its rows and drops its LanceDB table. Uploaded files in `shared_uploads/` are not deleted. --- ## 10. Development & Extension -### 10.1 Architecture Principles -- **Modular Design**: Clear separation of concerns -- **Configuration-Driven**: Behavior controlled via config files -- **Lazy Loading**: Components loaded on-demand -- **Thread Safety**: Proper synchronization for concurrent access +### 10.1 Principles + +* Configuration-driven: profiles in `rag_system/main.py`, construction in `rag_system/factory.py`. +* Lazy loading: embedders, rerankers and the pruner are built on first use and cached on the pipeline instance. +* One factory, one RAG API server, one owner per store. + +### 10.2 Extension points + +| To addโ€ฆ | Do this | +|---------|---------| +| A retriever | Implement the duck-typed contract `retrieve(text_query: str, table_name: str, k: int, search_type: str = "hybrid") -> List[Dict]` and return it from `RetrievalPipeline._get_dense_retriever()`. There is no `BaseRetriever` ABC. | +| A reranker | Plug it into `RetrievalPipeline._get_ai_reranker()`; a `rerankers`-library model only needs `reranker.model_name` + `reranker.model_type`. | +| A chunker | Add a `chunker_mode` branch in `IndexingPipeline.__init__`. | +| An embedding model | Point `EMBEDDING_MODEL` at a HuggingFace repo (contains `/`) or an Ollama tag, then re-index. | +| A pipeline profile | Add an entry to `PIPELINE_CONFIGS` and select it with `RAG_CONFIG_MODE` or `--mode`. | + +### 10.3 Validation + +There is no automated test suite in this repository. What exists: + +* `python system_health_check.py` โ€” imports, configuration dump, LanceDB connectivity, agent construction, embedding dimension, and a sample query against the first available table. +* `python run_system.py --health` โ€” HTTP health probes of all four services. +* `./test_docker_build.sh` โ€” builds the images and probes the container health endpoints. + +Building the automated tests is tracked in [`improvement_plan.md`](improvement_plan.md) ยง8. + +--- -### 10.2 Extension Points -- **Custom Retrievers**: Implement `BaseRetriever` interface -- **Custom Chunkers**: Extend chunking strategies -- **Custom Models**: Add new embedding or generation models -- **Custom Pipelines**: Create specialized processing workflows +## 11. Known Limitations -### 10.3 Testing Strategy -- **Unit Tests**: Individual component testing -- **Integration Tests**: End-to-end workflow testing -- **Performance Tests**: Load and stress testing -- **Health Checks**: Automated system validation +1. **Streamed chat turns are persisted via a follow-up call.** The stream itself (`POST :8001/chat/stream`) writes nothing to SQLite; the UI posts the completed turn to `POST :8000/sessions/{id}/messages/save` when the stream finishes. Direct stream consumers must do the same to get history. +2. **The RAG API serializes requests** โ€” one chat or indexing run at a time per process. Per-request option overrides are scoped to the request: the agent snapshots its config before applying them and restores it afterwards, so they no longer leak into subsequent requests. +3. **`enable_latechunk` defaults to `false` on `POST :8001/index`**, so an HTTP index build without that flag produces no late-chunk table even though the `default` profile enables late chunking. The CLI (`python -m rag_system.main index`) uses the profile value. +4. **A reranker that fails to load is skipped**, not replaced โ€” there is no fallback reranker. +5. **`requirements-docker.txt` has drifted** from `requirements.txt` and still lists packages with no importers. --- -> **Note**: This overview reflects the current implementation as of 2025-01-09. For the latest changes, check the git history and individual component documentation. \ No newline at end of file +> This overview describes the implementation as of 2026-08-08. When behaviour changes, update [`architecture_overview.md`](architecture_overview.md) and this file together. diff --git a/Documentation/triage_system.md b/Documentation/triage_system.md index bed44d4b..3932eb47 100644 --- a/Documentation/triage_system.md +++ b/Documentation/triage_system.md @@ -1,60 +1,101 @@ # ๐Ÿ”€ Triage / Routing System -_Maps to `rag_system/agent/loop.Agent._should_use_rag`, `_route_using_overviews`, and the fast-path router in `backend/server.py`._ +_One deterministic gate and one LLM router, in two processes:_ +* _`should_use_rag()` in `backend/server.py` (gateway, port 8000) โ€” deterministic, no LLM call; see "Backend gate" below._ +* _`Agent._triage_query_async` in `rag_system/agent/loop.py` (RAG API, port 8001) โ€” the only LLM routing layer._ ## Purpose -Determine, for every incoming query, whether it should be answered by: -1. **Direct LLM Generation** (no retrieval) โ€” faster, cheaper. -2. **Retrieval-Augmented Generation (RAG)** โ€” when the answer likely requires document context. - -## Decision Signals -| Signal | Source | Notes | -|--------|--------|-------| -| Keyword/regex check | `backend/server.py` (fast path) | Hard-coded quick wins (`what time`, `define`, etc.). | -| Index presence | SQLite (session โ†’ indexes) | If no indexes linked, direct LLM. | -| Overview routing | `_route_using_overviews()` | Uses document overviews and enrichment model to predict relevance. | -| LLM router prompt | `agent/loop.py` lines 648-665 | Final arbitrator (Ollama call, JSON output). | - -## High-level Flow +Decide, per query, whether to answer with: +1. **Direct LLM generation** โ€” no retrieval, faster and cheaper; or +2. **Retrieval-Augmented Generation** โ€” search the indexed documents first. + +## Which router actually runs + +| Request path | Router(s) involved | +|--------------|--------------------| +| Streaming chat (UI default): browser โ†’ `POST :8001/chat/stream` (`src/lib/api.ts:509`, toggle at `session-chat.tsx:48`, default on) | Agent router only. The backend gateway is not in this path. | +| Non-streaming chat: browser โ†’ `POST :8000/sessions//messages` โ†’ `POST :8001/chat` | Backend router first (`server.py:382`), then the agent router again inside the RAG API. | +| `POST :8000/chat` (`handle_chat`) | Neither. That endpoint always calls Ollama directly. | + +On the non-streaming path the backend gate decides `use_rag` locally; when it routes to RAG it forwards the query (and `force_rag`, when set) to the RAG API, where the agent triages again and may still choose `direct_answer`. Over-sending to RAG is therefore safe. + +## Agent router (`rag_system/agent/loop.py`) + +Order of evaluation in `_triage_query_async` (`loop.py:175-223`): + +1. **Overview routing** โ€” `_route_via_overviews(query)` (`loop.py:602-644`). Returns `None` immediately when no overviews are loaded (`loop.py:605-607`); otherwise it builds a `DOCUMENT OVERVIEWS:` block from the first 40 loaded overviews (`loop.py:612-613`), interpolates it into the router prompt (`loop.py:615-630`) and calls the utility model with `format="json"`. Parses `{"category": ...}`, defaulting to `rag_query` on a parse failure. +2. **History short-circuit** โ€” if the overview router returned `None` **and** the session already has chat history, the query is treated as a follow-up and routed to `rag_query` without any LLM call (`loop.py:188-193`). +3. **LLM fallback triage** โ€” a two-way classifier (`rag_query` / `direct_answer`) on the utility model, defaulting to `rag_query` if the JSON cannot be parsed. `Agent._normalize_triage()` runs on every verdict and collapses anything that is not an explicit `direct_answer` to `rag_query`, so a small model that emits the retired `graph_query` label still lands on the RAG path. + +`force_rag=true` โ€” or a compiled metadata `filters` object on the request โ€” skips all three: `query_type` is pinned to `rag_query` (`if force_rag or compiled_filters is not None` in `_run_async_inner`) while the `verify` / `ai_rerank` / `query_decompose` / `compose_sub_answers` / `context_expand` toggles all still apply. + +Both LLM routing calls run at `temperature: 0` (deterministic routing). + +There is no regex or keyword stage in the agent. + +## Backend gate (`backend/server.py`) + +Since the Phase-2 routing change (see `eval/decisions/phase2-gateway.md`), the gateway makes **no LLM call and reads no files** to route. `should_use_rag()` (module-level, unit-tested in `backend/test_gateway_routing.py`) evaluates in order: + +1. **`force_rag`** โ‡’ RAG, unconditionally (also forwarded, so agent triage is skipped too). +2. **No indexes linked** to the session โ‡’ direct LLM (nothing to retrieve from). +3. **Smalltalk / assistant-meta** โ€” a whole-message anchored allowlist (greetings, thanks, goodbyes, "who are you?"-style meta) capped at ~6 words โ‡’ direct LLM. +4. **Everything else** โ‡’ RAG. + +The old per-message enrichment-model router (`_route_using_overviews`) and the keyword/length fallback (`_simple_pattern_routing`) were deleted โ€” the fallback's substring matching misrouted most real document questions (`'hi'` matched *this* and *machine*). Over-sending to RAG is safe because the agent-side triage above can still answer directly; the gateway gate exists only to skip obvious non-retrieval turns at zero cost (~750 ms saved per routed message). + +## Flow + ```mermaid flowchart TD - Q["Incoming Query"] --> S1{Session\nHas Indexes?} - S1 -- no --> LLM["Direct LLM Generation"] - S1 -- yes --> S2{Fast Regex\nHeuristics} - S2 -- match--> LLM - S2 -- no --> S3{Overview\nRelevance > ฯ„?} - S3 -- low --> LLM - S3 -- high --> S4[LLM Router\n(prompt @648)] - S4 -- "route: RAG" --> RAG["Retrieval Pipeline"] - S4 -- "route: DIRECT" --> LLM + Q["Incoming query"] --> FR{force_rag or filters?} + FR -- yes --> RAG["Retrieval pipeline"] + FR -- no --> OV{Overviews loaded?} + OV -- no --> H{Chat history?} + OV -- yes --> R1["Overview router LLM
(utility model, JSON)"] + R1 -- rag_query --> RAG + R1 -- direct_answer --> LLM["Direct LLM answer"] + H -- yes --> RAG + H -- no --> R2["Fallback triage LLM
(rag_query / direct_answer)"] + R2 -- rag_query --> RAG + R2 -- direct_answer --> LLM ``` -## Detailed Sequence (Code-level) -1. **backend/server.py** - * `handle_session_chat()` builds `router_prompt` (line ~435) and makes a **first pass** decision before calling the heavy agent code. -2. **agent.loop._should_use_rag()** - * Re-evaluates using richer features (e.g., token count, query type). -3. **Overviews Phase** (`_route_using_overviews()`) - * Loads JSONL overviews file per index. - * Calls enrichment model (`qwen3:0.6b`) with prompt: _"Does this overview mention โ€ฆ ? "_ โ†’ returns yes/no. -4. **LLM Router** (prompt lines 648-665) - * JSON-only response `{ "route": "RAG" | "DIRECT" }`. - -## Interfaces & Dependencies -| Component | Calls / Data | -|-----------|--------------| -| SQLite `chat_sessions` | Reads `indexes` column to know linked index IDs. | -| LanceDB Overviews | Reads `index_store/overviews/.jsonl`. | -| `OllamaClient` | Generates LLM router decision. | - -## Config Flags -* `PIPELINE_CONFIGS.triage.enabled` โ€“ global toggle. -* Env var `TRIAGE_OVERVIEW_THRESHOLD` โ€“ min similarity score to prefer RAG (default 0.35). - -## Failure / Fallback Modes -1. If overview file missing โ†’ skip to LLM router. -2. If LLM router errors โ†’ default to RAG (safer) but log warning. +The backend gate is not in this diagram: it is a deterministic pre-filter (force_rag โ†’ indexes โ†’ smalltalk allowlist) with no LLM call, described above. + +## Overviews: where they come from + +| Step | Code | +|------|------| +| Written at index time, one JSON line per document: `{"doc_id": ..., "overview": ...}` | `rag_system/indexing/overview_builder.py:33-49` | +| Default file `index_store/overviews/overviews.jsonl`; the RAG API overrides it to `index_store/overviews/.jsonl` | `overview_builder.py:24`, `api_server.py:233-234` | +| Loaded per request by the RAG API before the agent runs | `api_server.py:362-366` โ†’ `Agent.load_overviews_for_indexes` (`loop.py:79-107`) | +| Falls back to the global `overviews.jsonl` when no per-index file exists | `loop.py:104-107` | + +If no overview file exists for the session, the agent's overview router returns `None` and routing falls through to the history short-circuit or the fallback triage prompt. (The backend gate does not read overview files at all.) + +## Models + +Only the agent-side router costs an LLM call, on the utility model โ€” `Agent._utility_model()` resolves `ENRICHMENT_MODEL` env var โ†’ `OLLAMA_CONFIG["enrichment_model"]` โ†’ `qwen3.5:4b`. The gateway gate is pure Python. Routing is never charged to the generation model, and a per-request `model` override does not change the routing model: the RAG API applies that override only for the duration of the request via a context manager, and the router reads `enrichment_model`, not `generation_model`. + +## Configuration + +| Knob | Where | Effect | +|------|-------|--------| +| `force_rag` (`forceRag`) | request body on `/chat`, `/chat/stream` (`api_server.py:191`) and on the backend's `/sessions//messages` (`server.py:381`) | Skips triage entirely and forces the RAG path. Surfaced in the UI as the "Always search documents" toggle (`session-chat.tsx:51`, default off). | + +The third outcome, `graph_query`, and the `graph_strategy` config block that armed it were **removed on 2026-08-09** (roadmap item 2.5) along with the rest of the graph module. Evidence: GraphRAG loses on single-hop retrieval, its multi-hop gains are contested, and it costs 41โ€“57ร— at indexing and up to ~377ร— in query tokens โ€” [`research/academic-evidence-2026.md`](research/academic-evidence-2026.md) ยง6. + +There is no global triage on/off switch and no similarity threshold. `PIPELINE_CONFIGS` has no `triage` key, and no `TRIAGE_OVERVIEW_THRESHOLD` environment variable is read anywhere. + +## Failure / fallback modes + +| Failure | Agent | Backend | +|---------|-------|---------| +| No overviews on disk | `_route_via_overviews` returns `None`; history short-circuit or fallback triage decides | n/a โ€” gateway gate reads no files | +| Router LLM returns unparseable JSON / unexpected text | defaults to `rag_query` | n/a โ€” gateway gate makes no LLM call | +| Router LLM call fails (timeout / connection / bad status) | the client catches the request error and returns `{}`, so the unparseable-JSON default applies โ€” triage fails closed to `rag_query` | n/a | --- -_Keep this document updated whenever routing heuristics, thresholds, or prompt wording change._ \ No newline at end of file +_Keep this document updated whenever routing order, prompts, or fallback behaviour change._ diff --git a/Documentation/verifier.md b/Documentation/verifier.md index a1c5bf7d..151401ab 100644 --- a/Documentation/verifier.md +++ b/Documentation/verifier.md @@ -1,49 +1,128 @@ # โœ… Answer Verifier -_File: `rag_system/agent/verifier.py`_ +_File: `rag_system/agent/verifier.py`. Sole caller: `rag_system/agent/loop.py:560-578`._ ## Objective -Assess whether an answer produced by RAG is **grounded** in the retrieved context snippets. +Assess whether an answer produced by the RAG path is **grounded** in the retrieved context snippets, and annotate the answer with the model's self-reported confidence. + +Two interchangeable backends implement it. The **LLM-prompt verifier below is what +ships**; a local NLI/verifier model is opt-in via `VERIFIER_MODEL` (see +[Local verifier model](#local-verifier-model-opt-in)). + +> **`[Confidence: N%]` is UX, not a measurement.** It is whatever the verifier +> emitted, rescaled to a percent. Neither backend is calibrated: an 80% does not +> mean the answer is right four times in five. Swapping the LLM prompt for an NLI +> model changes where the number comes from, not that caveat. + +## Prompt +See `prompt_inventory.md` โ†’ `verifier.fact_check` (`verifier.py:25-85`). The prompt carries three few-shot examples and then a `# TASK` block into which the query, the context (clamped to the first 4000 characters at `verifier.py:76`) and the answer are injected. It is sent asynchronously with `format="json"` at `temperature: 0` โ€” deterministic verdicts, the same pin the eval judge got. + +Expected response, one line of JSON: -## Prompt (see `prompt_inventory.md` `verifier.fact_check`) -Strict JSON schema: ```jsonc { "verdict": "SUPPORTED" | "NOT_SUPPORTED" | "NEEDS_CLARIFICATION", "is_grounded": true | false, - "reasoning": "< โ‰ค30 words >", + "reasoning": "", "confidence_score": 0-100 } ``` -## Sequence Diagram +It is parsed into a `VerificationResult` (`verifier.py:4-9`) with those four fields. + +## Sequence + ```mermaid sequenceDiagram - participant RP as Retrieval Pipeline + participant A as Agent._run_async participant V as Verifier - participant LLM as Ollama + participant LLM as Ollama (utility model) - RP->>V: query, context, answer - V->>LLM: verification prompt + A->>A: build context_str from result["source_documents"] + A->>V: verify_async(contextual_query, context_str, answer) + V->>LLM: fact-check prompt (format=json) LLM-->>V: JSON verdict - V-->>RP: VerificationResult + V-->>A: VerificationResult + A->>A: append confidence tag to result["answer"] ``` -## Usage Sites -| Caller | Code | When | -|--------|------|------| -| `RetrievalPipeline.answer_stream()` | `pipelines/retrieval_pipeline.py` | If `verify=true` flag from frontend. | -| `Agent.loop.run()` | fallback path | Experimental for composed answers. | +## Call site + +| Caller | Code | When it runs | +|--------|------|--------------| +| `Agent._run_async()` | `rag_system/agent/loop.py`, end of `_run_async` | After every branch (direct answer, decomposed/composed, single-query RAG), when verification is enabled **and** `result["source_documents"]` is non-empty. | + +There is exactly one call site in the repository. `rag_system/pipelines/retrieval_pipeline.py` does not import or reference `Verifier`. Only the async `verify_async()` exists โ€” the synchronous `verify()` was removed (`verifier.py:20`). + +Because the check is gated on non-empty `source_documents`, the `direct_answer` route (which returns `source_documents: []`) is never verified. + +## Configuration + +| Knob | Where | Default | Meaning | +|------|-------|---------|---------| +| `verification.enabled` | `rag_system/main.py:80` (`default` profile) | `true` | Profile-level switch. | +| `verification.enabled` | `rag_system/main.py:109` (`fast` profile) | `false` | Verification off in the speed profile. | +| โ€” | `loop.py:560` | `true` | Fallback used when the profile has no `verification` block. | +| `verify` | HTTP request field on `/chat` and `/chat/stream` (`api_server.py:186`) | not sent โ‡’ profile value wins | Per-request override; forwarded to `Agent.run(verify=...)`. Also accepted by the backend gateway as `verify` (`backend/server.py:48`). | +| model | `loop.py`, `Agent.__init__` | utility model (`enrichment_model`, default `qwen3.5:4b`) | Which Ollama model runs the LLM-prompt verifier. Verification runs on the small model, not the answer model. | +| `verification.model` / `VERIFIER_MODEL` | pipeline config, or the env var | unset โ‡’ LLM-prompt verifier | A HuggingFace model name switches the backend to a local NLI/verifier model. | +| `verification.threshold` | pipeline config | `0.5` | Score at or above which the local verifier calls an answer grounded. Ignored by the LLM-prompt backend. | +| `VERIFIER_TRUST_REMOTE_CODE` | env var | unset | Must be `1` to load a verifier that ships custom modelling code (e.g. Vectara HHEM). | + +## Local verifier model (opt-in) + +_Roadmap item 2.4, shipped 2026-08-09 as a **seam**: the default is unchanged._ + +```bash +VERIFIER_MODEL=MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli python -m rag_system.main api +``` + +`LocalNLIVerifier` (`rag_system/agent/verifier.py`) loads any HuggingFace +sequence-classification model **lazily on first use**, splits the answer into +sentences, scores each one against the retrieved evidence as the premise, and +takes the **minimum** โ€” one unsupported sentence makes the answer ungrounded, +matching the binary semantics `eval/judge.py` already uses. The "supported" logit +is resolved from `id2label` (`entailment` / `consistent` / `supported` / `1`), +falling back to the last class for binary checkers. + +A model that cannot be loaded **raises** with the list of names that were +checked; it does not silently fall back to the LLM prompt. A verifier that +quietly is not the verifier you configured is worse than an error. + +### Availability, checked 2026-08-09 + +| Candidate | Verdict | +|---|---| +| **ThinknCheck** (arXiv 2604.01652, UPenn, 1B, 78.1 BAcc) | **No public weights.** The paper is real, but a HuggingFace Hub search for `thinkncheck` returns zero models and the paper links no release. Cannot be wired. | +| `ibm-granite/granite-guardian-3.3-8b` | Exists, Apache-2.0 โ€” but 8B / ~16 GB, far over the budget this seam is for. | +| `ibm-granite/granite-guardian-hap-38m` | Exists, 38M, Apache-2.0 โ€” but it is a **hate/abuse/profanity** RoBERTa classifier. Wrong task: it does not score answer-vs-evidence entailment. | +| `MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli` | โœ… MIT, 369 MB, no custom code. Generic NLI. | +| `lytang/MiniCheck-DeBERTa-v3-Large` | โœ… MIT, 1.74 GB, no custom code. Purpose-built grounded claim verification (the baseline ThinknCheck benchmarks against). | +| `vectara/hallucination_evaluation_model` (HHEM-2.1-open) | Apache-2.0, 438 MB, but ships custom modelling code โ€” needs `VERIFIER_TRUST_REMOTE_CODE=1`. | + +The same table is embedded in the code as `VERIFIER_AVAILABILITY_NOTES` and is +printed verbatim when a configured verifier fails to load. + +The UI initialises its verify toggle to `true` (`src/components/ui/session-chat.tsx:49`), so verification is on by default for chat traffic. + +## Effect on the answer + +The verifier does **not** add a field to the response. It mutates the answer string (`loop.py:568-578`): + +* `confidence_score > 0` โ†’ appends `" [Confidence: N%]"`. +* Additionally, when `is_grounded` is false **or** the score is below 50 โ†’ appends `" [Warning: Low confidence. Groundedness: ]"`. +* `confidence_score == 0` โ†’ nothing is appended (0 is treated as a parse failure) and a warning is logged to stdout. + +The API response shape is unchanged: `{"answer": ..., "source_documents": [...]}`. + +## Failure modes + +* Invalid JSON, a missing `response` key, or a type-mismatched verdict (a string `"85"`, `null`, the string `"false"`) โ†’ the parse/coercion in `verify_async()` fails open to `VerificationResult(False, 0)`, and because the score is 0 no tag is appended โ€” the answer is returned unannotated. Malformed verdict JSON degrades to score 0; it cannot 500. +* HTTP-layer failures of the LLM call itself (timeout, connection error, non-200 status) are caught inside `generate_completion_async`, which returns `{}` โ†’ the same score-0 path, so the answer comes back unannotated with HTTP 200. Only **non-httpx** exceptions (e.g. `VerifierModelUnavailable` from the local-verifier seam) propagate out of `_run_async` to the API handler, which returns a 500 (or an SSE `error` event on the streaming endpoint). There is no try/except around the `verify_async` call itself. -## Config -| Flag | Default | Meaning | -|------|---------|---------| -| `verify` | false | Frontend toggle; if true verifier runs. | -| `generation_model` | `qwen3:8b` | Same model as answer generation. +## Cost -## Failure Modes -* If LLM returns invalid JSON โ†’ parse exception handled, result = NOT_SUPPORTED. -* If verification call times out โ†’ pipeline logs but still returns answer (unverified). +Verification is one extra LLM round-trip per answered query, on the utility model, with a prompt containing up to 4000 characters of context. Set `verify: false` on the request, or run the `fast` profile, to skip it. --- -_Keep updated when schema or usage flags change._ \ No newline at end of file +_Keep updated when the schema, the gating conditions, or the answer annotations change._ diff --git a/README.md b/README.md index 702dcd0d..af47dc62 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@ [![GitHub Forks](https://img.shields.io/github/forks/PromtEngineer/localGPT?style=flat-square)](https://github.com/PromtEngineer/localGPT/network/members) [![GitHub Issues](https://img.shields.io/github/issues/PromtEngineer/localGPT?style=flat-square)](https://github.com/PromtEngineer/localGPT/issues) [![GitHub Pull Requests](https://img.shields.io/github/issues-pr/PromtEngineer/localGPT?style=flat-square)](https://github.com/PromtEngineer/localGPT/pulls) -[![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg?style=flat-square)](https://www.python.org/downloads/) +[![Python 3.10+](https://img.shields.io/badge/python-3.10+-blue.svg?style=flat-square)](https://www.python.org/downloads/) [![License](https://img.shields.io/badge/license-MIT-green.svg?style=flat-square)](LICENSE) [![Docker](https://img.shields.io/badge/docker-supported-blue.svg?style=flat-square)](https://www.docker.com/) @@ -28,12 +28,12 @@ LocalGPT is a **fully private, on-premise Document Intelligence platform**. Ask questions, summarise, and uncover insights from your files with state-of-the-art AIโ€”no data ever leaves your machine. -More than a traditional RAG (Retrieval-Augmented Generation) tool, LocalGPT features a **hybrid search engine** that blends semantic similarity, keyword matching, and [Late Chunking](https://jina.ai/news/late-chunking-in-long-context-embedding-models/) for long-context precision. A **smart router** automatically selects between RAG and direct LLM answering for every query, while **contextual enrichment** and sentence-level [Context Pruning](https://huggingface.co/naver/provence-reranker-debertav3-v1) surface only the most relevant content. An independent **verification** pass adds an extra layer of accuracy. +More than a traditional RAG (Retrieval-Augmented Generation) tool, LocalGPT features a **hybrid search engine** that fuses dense vector search with LanceDB's native full-text search, arbitrated by a **calibrated cross-encoder reranker**. A **smart router** picks between RAG and direct LLM answering for every query, while **contextual enrichment** and sentence-level [Context Pruning](https://huggingface.co/naver/provence-reranker-debertav3-v1) surface only the most relevant content. Optional passes โ€” [Late Chunking](https://jina.ai/news/late-chunking-in-long-context-embedding-models/), an independent answer **verification** step, and experimental multi-vector (late-interaction) retrieval โ€” can be switched on per config; the defaults ship with exactly the components that earned their place in measured evaluations (see [`eval/decisions/`](eval/decisions/)). -The architecture is **modular and lightweight**โ€”enable only the components you need. With a pure-Python core and minimal dependencies, LocalGPT is simple to deploy, run, and maintain on any infrastructure.The system has minimal dependencies on frameworks and libraries, making it easy to deploy and maintain. The RAG system is pure python and does not require any additional dependencies. +The architecture is **modular and lightweight**โ€”enable only the components you need. The RAG core is plain Python built on the standard library's HTTP server, with no web framework and no agent framework in the way. ## โ–ถ๏ธ Video -Watch this [video](https://youtu.be/JTbtGH3secI) to get started with LocalGPT. +Watch this [video](https://youtu.be/JTbtGH3secI) to get started with LocalGPT. | Home | Create Index | Chat | |------|--------------|------| @@ -42,73 +42,64 @@ Watch this [video](https://youtu.be/JTbtGH3secI) to get started with LocalGPT. ## โœจ Features - **Utmost Privacy**: Your data remains on your computer, ensuring 100% security. -- **Versatile Model Support**: Seamlessly integrate a variety of open-source models via Ollama. -- **Diverse Embeddings**: Choose from a range of open-source embeddings. +- **Versatile Model Support**: Swap generation models freely via Ollama. +- **Diverse Embeddings**: HuggingFace embedding models (harrier-oss-v1, the Qwen3-Embedding family) or any Ollama embedding tag. - **Reuse Your LLM**: Once downloaded, reuse your LLM without the need for repeated downloads. -- **Chat History**: Remembers your previous conversations (in a session). -- **API**: LocalGPT has an API that you can use for building RAG Applications. -- **GPU, CPU, HPU & MPS Support**: Supports multiple platforms out of the box, Chat with your data using `CUDA`, `CPU`, `HPU (Intelยฎ Gaudiยฎ)` or `MPS` and more! +- **API**: A REST gateway on port 8000 and the RAG API on port 8001 for building your own applications. +- **CUDA, MPS & CPU**: Embedding and reranking pick CUDA, then Apple MPS, then CPU automatically. ### ๐Ÿ“– Document Processing -- **Multi-format Support**: PDF, DOCX, TXT, Markdown, and more (Currently only PDF is supported) -- **Contextual Enrichment**: Enhanced document understanding with AI-generated context, inspired by [Contextual Retrieval](https://www.anthropic.com/news/contextual-retrieval) -- **Batch Processing**: Handle multiple documents simultaneously +- **Formats**: PDF, DOCX, HTML/HTM, Markdown, and TXT, parsed by [Docling](https://github.com/docling-project/docling) +- **OCR fallback**: PDFs with no text layer are re-run through Docling's OCR pipeline; the engine is chosen from whatever is installed (OcrMac on macOS, then EasyOCR, RapidOCR, tesserocr, or the `tesseract` CLI) +- **Contextual Enrichment**: Chunk-level context generated by a small LLM, inspired by [Contextual Retrieval](https://www.anthropic.com/news/contextual-retrieval) +- **Late Chunking** (off by default): A second, document-level embedding pass stored in a companion `
_lc` table. The 2026-08-18 component ablation measured its removal at the noise floor on single-turn quality while it doubles the vectors written per index, so it now ships disabled; one flag (`retrieval.latechunk.enabled`) re-enables both the index-time build and the query-time leg โ€” multi-turn conversations with drifting phrasing are where it earns its cost +- **Document Overviews**: A short per-document summary written to `index_store/overviews/.jsonl` and used by the router ### ๐Ÿค– AI-Powered Chat - **Natural Language Queries**: Ask questions in plain English -- **Source Attribution**: Every answer includes document references -- **Smart Routing**: Automatically chooses between RAG and direct LLM responses -- **Query Decomposition**: Breaks complex queries into sub-questions for better answers -- **Semantic Caching**: TTL-based caching with similarity matching for faster responses -- **Session-Aware History**: Maintains conversation context across interactions -- **Answer Verification**: Independent verification pass for accuracy -- **Multiple AI Models**: Ollama for inference, HuggingFace for embeddings and reranking - +- **Source Attribution**: Answers come back with the chunks they were grounded in +- **Smart Routing**: Chooses between RAG and a direct LLM answer per query +- **Query Decomposition**: Splits complex questions into sub-questions, retrieves per sub-question, then pools the candidates for one rerank and one synthesis pass (per-sub-answer composition is available as an option) +- **Reciprocal Rank Fusion**: Vector and full-text hits are fused with RRF โ€” no weights to tune +- **Reranking**: A cross-encoder pass over the fused candidate set, on by default with calibrated score-based selection ([`eval/DECISIONS.md`](eval/DECISIONS.md)) +- **Sentence Pruning**: Optional Provence pruning drops irrelevant sentences from each chunk +- **Semantic Caching**: TTL cache with a 0.98 similarity threshold, scoped to the session +- **Answer Verification** (off by default): A second pass that appends `[Confidence: N%]` to the answer. Ablation measured zero verdict flips from disabling it โ€” it annotates rather than changes answers โ€” so it ships disabled; re-enable with `verification.enabled` ### ๐Ÿ› ๏ธ Developer-Friendly -- **RESTful APIs**: Complete API access for integration -- **Real-time Progress**: Live updates during document processing -- **Flexible Configuration**: Customize models, chunk sizes, and search parameters -- **Extensible Architecture**: Plugin system for custom components +- **RESTful APIs**: Every UI action is a documented HTTP call +- **Streaming phases**: Server-Sent Events expose each pipeline stage as it runs +- **Flexible Configuration**: Models, chunk size, retrieval mode and toggles per request +- **One master config**: `rag_system/main.py` holds every default, overridable by environment variable ### ๐ŸŽจ Modern Interface - **Intuitive Web UI**: Clean, responsive design - **Session Management**: Organize conversations by topic - **Index Management**: Easy document collection management -- **Real-time Chat**: Streaming responses for immediate feedback +- **Live Progress**: Retrieval, reranking and synthesis stages stream into the chat as they happen --- ## ๐Ÿš€ Quick Start -Note: The installation is currently only tested on macOS. - ### Prerequisites -- Python 3.8 or higher (tested with Python 3.11.5) -- Node.js 16+ and npm (tested with Node.js 23.10.0, npm 10.9.2) +- Python 3.10+ (3.11 recommended โ€” the Docker images use `python:3.11-slim`) +- Node.js 20+ and npm - Docker (optional, for containerized deployment) - 8GB+ RAM (16GB+ recommended) - Ollama (required for both deployment approaches) -### ***NOTE*** -Before this brach is moved to the main branch, please clone this branch for instalation: - -```bash -git clone -b localgpt-v2 https://github.com/PromtEngineer/localGPT.git -cd localGPT -``` - -### Option 1: Docker Deployment +### Option 1: Docker Deployment ```bash # Clone the repository git clone https://github.com/PromtEngineer/localGPT.git cd localGPT -# Install Ollama locally (required even for Docker) +# Install Ollama locally (recommended even for Docker) curl -fsSL https://ollama.ai/install.sh | sh -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b # Start Ollama ollama serve @@ -120,6 +111,19 @@ ollama serve open http://localhost:3000 ``` +If you would rather not install Ollama on the host, run it as a container instead: + +```bash +./start-docker.sh container +# then pull the models inside the container +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` + +`./start-docker.sh` (with no argument) uses local Ollama. If nothing is listening on +port 11434 it offers to switch to the containerized Ollama; add `-y` (or set +`NONINTERACTIVE=1`) to take that fallback without a prompt in scripts and CI. + **Docker Management Commands:** ```bash # Check container status @@ -143,20 +147,18 @@ cd localGPT pip install -r requirements.txt # Key dependencies installed: -# - torch==2.4.1, transformers==4.51.0 (AI models) -# - lancedb (vector database) -# - rank_bm25, fuzzywuzzy (search algorithms) -# - sentence_transformers, rerankers (embedding/reranking) -# - docling (document processing) -# - colpali-engine (multimodal processing - support coming soon) +# - torch==2.4.1, transformers==4.51.0 (embedding + reranker models) +# - lancedb (vector store and full-text search) +# - rerankers (cross-encoder reranking) +# - docling (document parsing) # Install Node.js dependencies npm install # Install and start Ollama curl -fsSL https://ollama.ai/install.sh | sh -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ollama serve # Start the system (in a new terminal) @@ -168,32 +170,35 @@ open http://localhost:3000 **System Management:** ```bash -# Check system health (comprehensive diagnostics) +# Check system health (loads the models and runs a sample query) python system_health_check.py -# Check service status and health +# Real HTTP health checks against each service; exits non-zero if one is unhealthy python run_system.py --health -# Start in production mode +# Start in production mode (runs `npm run build` before `next start`) python run_system.py --mode prod -# Skip frontend (backend + RAG API only) +# Skip frontend (Ollama + RAG API + backend only) python run_system.py --no-frontend -# View aggregated logs +# Tail logs/*.log from another shell python run_system.py --logs-only -# Stop all services +# Stop everything recorded in logs/run_system.pid python run_system.py --stop # Or press Ctrl+C in the terminal running python run_system.py ``` **Service Architecture:** -The `run_system.py` launcher manages four key services: -- **Ollama Server** (port 11434): AI model serving -- **RAG API Server** (port 8001): Document processing and retrieval -- **Backend Server** (port 8000): Session management and API endpoints -- **Frontend Server** (port 3000): React/Next.js web interface +The `run_system.py` launcher manages four services and writes their PIDs to `logs/run_system.pid`: +- **Ollama Server** (port 11434): model serving โ€” reused if already running +- **RAG API Server** (port 8001): indexing, retrieval and the agent loop +- **Backend Server** (port 8000): sessions, indexes, uploads, chat history +- **Frontend Server** (port 3000): Next.js web interface (optional โ€” skipped if `npm` is missing) + +On startup the launcher checks that `qwen3.5:9b` and `qwen3.5:4b` are present and +runs `ollama pull` for anything missing. ### Option 3: Manual Component Startup @@ -203,9 +208,10 @@ ollama serve # Terminal 2: Start RAG API python -m rag_system.api_server +# equivalently: python -m rag_system.main api --port 8001 # Terminal 3: Start Backend -cd backend && python server.py +python backend/server.py # Terminal 4: Start Frontend npm run dev @@ -213,6 +219,11 @@ npm run dev # Access at http://localhost:3000 ``` +> Run every command from the repository root. Relative paths (`backend/chat_data.db`, +> `lancedb/`, `index_store/`, `shared_uploads/`) resolve against the current working +> directory, so `cd backend && python server.py` would create a second database at +> `backend/backend/chat_data.db`. + --- ### Detailed Installation @@ -222,62 +233,69 @@ npm run dev **Ubuntu/Debian:** ```bash sudo apt update -sudo apt install python3.8 python3-pip nodejs npm docker.io docker-compose +sudo apt install python3.11 python3-pip nodejs npm docker.io docker-compose-plugin ``` **macOS:** ```bash -brew install python@3.8 node npm docker docker-compose +brew install python@3.11 node docker ``` **Windows:** ```bash -# Install Python 3.8+, Node.js, and Docker Desktop +# Install Python 3.10+, Node.js 20+, and Docker Desktop # Then use PowerShell or WSL2 ``` #### 2. Install AI Models -**Install Ollama (Recommended):** +Only the two Ollama models need an explicit pull. The embedding model +(`microsoft/harrier-oss-v1-0.6b`, 1.2 GB) is downloaded from HuggingFace the +first time it is used; the reranker (~7.5 GB) is loaded lazily โ€” downloaded +on the first reranked query. + ```bash # Install Ollama curl -fsSL https://ollama.ai/install.sh | sh -# Pull recommended models -ollama pull qwen3:0.6b # Fast generation model -ollama pull qwen3:8b # High-quality generation model -``` - -#### 3. Configure Environment - -```bash -# Copy environment template -cp .env.example .env - -# Edit configuration -nano .env -``` - -**Key Configuration Options:** -```env -# AI Models (referenced in rag_system/main.py) -OLLAMA_HOST=http://localhost:11434 - -# Database Paths (used by backend and RAG system) -DATABASE_PATH=./backend/chat_data.db -VECTOR_DB_PATH=./lancedb - -# Server Settings (used by run_system.py) -BACKEND_PORT=8000 -FRONTEND_PORT=3000 -RAG_API_PORT=8001 - -# Optional: Override default models -GENERATION_MODEL=qwen3:8b -ENRICHMENT_MODEL=qwen3:0.6b -EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B -RERANKER_MODEL=answerdotai/answerai-colbert-small-v1 -``` +# Pull the default models +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, enrichment, verification +``` + +#### 3. Configure Environment (optional) + +Every setting has a working default, so LocalGPT runs with no `.env` at all. +To override one, create a `.env` in the repository root (`rag_system/main.py` +calls `load_dotenv()` at import, before its config constants are evaluated; the +factory calls it again defensively). `.env.example` lists the same variables +with their code defaults. + +| Variable | Default | Read by | +|----------|---------|---------| +| `OLLAMA_HOST` | `http://localhost:11434` | `rag_system/main.py`, `backend/ollama_client.py` | +| `RAG_API_URL` | `http://localhost:8001` | `backend/server.py` (builds `/chat` and `/index`) | +| `NEXT_PUBLIC_API_URL` | `http://localhost:8000` | `src/lib/api.ts` โ€” inlined at `npm run build` | +| `NEXT_PUBLIC_RAG_API_URL` | `http://localhost:8001` | `src/lib/api.ts` โ€” inlined at `npm run build` | +| `DB_PATH` | `backend/chat_data.db` | `backend/database.py` | +| `LANCEDB_PATH` | `storage.lancedb_uri` (`./lancedb`) | `rag_system/main.py` (pipeline profiles), `backend/database.py`, `system_health_check.py` | +| `GENERATION_MODEL` | `qwen3.5:9b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | `rag_system/main.py` | +| `RERANKER_MODEL` | `Qwen/Qwen3-Reranker-4B` (loaded lazily on the first reranked query) | `rag_system/main.py` | +| `RAG_CONFIG_MODE` | `default` | `rag_system/api_server.py` (`default` or `fast`) | +| `RAG_API_TIMEOUT` | `600` | `backend/server.py` (seconds to wait for a chat answer) | +| `RAG_API_INDEX_TIMEOUT` | `3600` | `backend/server.py` (seconds to wait for an indexing run) | +| `LLM_BACKEND` | `ollama` | `rag_system/main.py` (`ollama` or `watsonx`) | +| `HF_TOKEN` | unset | HuggingFace, for gated model downloads | + +`NEXT_PUBLIC_*` values are baked into the frontend bundle by `next build`. +Changing them requires a rebuild (`npm run build`, or `docker compose build frontend`). + +> **Changing `EMBEDDING_MODEL` invalidates existing indexes.** Vector width is read +> from the loaded model, and appending vectors of a different width to an existing +> LanceDB table raises an error telling you to rebuild. Re-create your indexes after +> switching embedding models. #### 4. Initialize the System @@ -285,13 +303,13 @@ RERANKER_MODEL=answerdotai/answerai-colbert-small-v1 # Run system health check python system_health_check.py -# Initialize databases +# Initialize the SQLite database python -c "from backend.database import ChatDatabase; ChatDatabase().init_database()" -# Test installation -python -c "from rag_system.main import get_agent; print('โœ… Installation successful!')" +# Test the RAG imports +python -c "from rag_system.factory import get_agent; print('โœ… Installation successful!')" -# Validate complete setup +# Validate the running services python run_system.py --health ``` @@ -306,32 +324,51 @@ An **index** is a collection of processed documents that you can chat with. #### Using the Web Interface: 1. Open http://localhost:3000 2. Click "Create New Index" -3. Upload your documents (PDF, DOCX, TXT) +3. Upload your documents (PDF, DOCX, TXT, MD, HTML) 4. Configure processing options 5. Click "Build Index" -#### Using Scripts: +#### Using the CLI: ```bash -# Simple script approach -./simple_create_index.sh "My Documents" "path/to/document.pdf" +# Index a single file or a whole directory with the 'default' profile +python -m rag_system.main index ./my_documents + +# Use the speed-optimised profile instead +python -m rag_system.main index ./my_documents --mode fast -# Interactive script +# Ask one question and print the JSON result +python -m rag_system.main chat "What are the key findings?" +``` + +`index` walks a directory for `.pdf`, `.docx`, `.html`, `.htm`, `.md` and `.txt` +files. It writes into the profile's `storage.text_table_name` (`text_pages_v4`), +which is *not* the per-index table the web UI creates. + +#### Using the interactive script (creates a UI-visible index): +```bash +# Guided prompts: name, documents, chunk size, models python create_index_script.py + +# Non-interactive, from a JSON file +python create_index_script.py --create-sample # writes index_config.sample.json +python create_index_script.py --batch index_config.sample.json ``` -#### Using API: +#### Using the HTTP API: ```bash # Create index curl -X POST http://localhost:8000/indexes \ -H "Content-Type: application/json" \ -d '{"name": "My Index", "description": "My documents"}' -# Upload documents +# Upload documents (form field name must be "files") curl -X POST http://localhost:8000/indexes/INDEX_ID/upload \ -F "files=@document.pdf" # Build index -curl -X POST http://localhost:8000/indexes/INDEX_ID/build +curl -X POST http://localhost:8000/indexes/INDEX_ID/build \ + -H "Content-Type: application/json" \ + -d '{"chunk_size": 512, "enable_enrich": true, "enable_latechunk": true}' ``` ### 2. Start Chatting @@ -345,96 +382,138 @@ Once your index is built: ### 3. Advanced Features -#### Custom Model Configuration +#### Per-session and per-request model choice ```bash -# Use different models for different tasks +# The session's default generation model curl -X POST http://localhost:8000/sessions \ -H "Content-Type: application/json" \ - -d '{ - "title": "High Quality Session", - "model": "qwen3:8b", - "embedding_model": "Qwen/Qwen3-Embedding-4B" - }' -``` + -d '{"title": "High Quality Session", "model": "qwen3.6:27b"}' -#### Batch Document Processing -```bash -# Process multiple documents at once -python demo_batch_indexing.py --config batch_indexing_config.json +# Override it for one message +curl -X POST http://localhost:8000/sessions/SESSION_ID/messages \ + -H "Content-Type: application/json" \ + -d '{"message": "Summarise section 3", "model": "qwen3.5:4b"}' ``` +The embedding model is a property of the index, not the session โ€” choose it when +you build the index. + #### API Integration ```python import requests -# Chat with your documents via API -response = requests.post('http://localhost:8000/chat', json={ +# Talk to the RAG API directly +response = requests.post('http://localhost:8001/chat', json={ 'query': 'What are the key findings in the research papers?', 'session_id': 'your-session-id', - 'search_type': 'hybrid', - 'retrieval_k': 20 + 'retrieval_mode': 'hybrid', + 'retrieval_k': 20, }) -print(response.json()['response']) +print(response.json()['answer']) ``` --- ## ๐Ÿ”ง Configuration +All defaults live in `rag_system/main.py`. Every model name there can be +overridden with the environment variables listed above. + ### Model Configuration -LocalGPT supports multiple AI model providers with centralized configuration: +| Role | Default | Documented options | +|------|---------|--------------------| +| Generation (answers) | `qwen3.5:9b` | `qwen3.6:27b` (high-end, ~17GB), `qwen3.5:4b` (light) | +| Enrichment / utility (routing, triage, decomposition, verification) | `qwen3.5:4b` | `qwen3.5:2b` (light) | +| Embedding | `microsoft/harrier-oss-v1-0.6b` (MIT, 1024 dims) | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context, for multilingual / long-context corpora), `Qwen/Qwen3-Embedding-0.6B` (1024 dims) | +| Reranker (on by default) | `Qwen/Qwen3-Reranker-4B` | `BAAI/bge-reranker-v2-m3` (low latency), `answerdotai/answerai-colbert-small-v1`, `Qwen/Qwen3-Reranker-0.6B` | -#### Ollama Models (Local Inference) ```python +# rag_system/main.py OLLAMA_CONFIG = { - "host": "http://localhost:11434", - "generation_model": "qwen3:8b", # Main text generation - "enrichment_model": "qwen3:0.6b" # Lightweight routing/enrichment + "host": os.getenv("OLLAMA_HOST", "http://localhost:11434"), + "generation_model": os.getenv("GENERATION_MODEL", "qwen3.5:9b"), + "enrichment_model": os.getenv("ENRICHMENT_MODEL", "qwen3.5:4b"), } -``` -#### External Models (HuggingFace Direct) -```python EXTERNAL_MODELS = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", # 1024 dimensions - "reranker_model": "answerdotai/answerai-colbert-small-v1", # ColBERT reranker - "fallback_reranker": "BAAI/bge-reranker-base" # Backup reranker + "embedding_model": os.getenv("EMBEDDING_MODEL", "microsoft/harrier-oss-v1-0.6b"), + "reranker_model": os.getenv("RERANKER_MODEL", "Qwen/Qwen3-Reranker-4B"), } ``` +Embedding dimensions are never hardcoded โ€” they are measured from the vectors the +loaded model produces. If the reranker fails to load, the pipeline logs a warning +and continues **without** reranking rather than falling back to another model. + +Vision / multimodal models are not part of the pipeline. PDF parsing and OCR are +handled by Docling. Models such as GLM-OCR or Qwen3-VL could be added as a +pre-processing step, but **they are not integrated today**. + ### Pipeline Configuration -LocalGPT offers two main pipeline configurations: +`PIPELINE_CONFIGS` has exactly two profiles. Select one with `RAG_CONFIG_MODE` +(RAG API) or `--mode` (CLI). #### Default Pipeline (Production-Ready) ```python "default": { - "description": "Production-ready pipeline with hybrid search, AI reranking, and verification", + "description": "Production-ready pipeline with hybrid search, query decomposition, and verification", "storage": { "lancedb_uri": "./lancedb", - "text_table_name": "text_pages_v3", - "bm25_path": "./index_store/bm25" + "text_table_name": "text_pages_v4" }, "retrieval": { - "retriever": "multivector", "search_type": "hybrid", - "late_chunking": {"enabled": True}, - "dense": {"enabled": True, "weight": 0.7}, - "bm25": {"enabled": True} + # Off since the 2026-08-18 component ablation; one flag covers the + # index-time build and the query-time leg. + "latechunk": {"enabled": False}, + "dense": {"enabled": True}, + "retry": {"enabled": True, "min_top_score": 0.12, "max_attempts": 1}, + # Phase-4 features, all off until benchmarked: + "document_escalation": {"enabled": False, "max_documents": 1, "token_budget": 6000}, + "crossref_hop": {"enabled": False, "max_hops": 1, "chunks_per_hop": 3}, + "overview_prefilter": {"enabled": False, "top_documents": 5, "mode": "boost"} }, + "embedding_model_name": EXTERNAL_MODELS["embedding_model"], + # On since arm G (2026-08-14): min_score keeps only candidates the + # calibrated Qwen scorer marks relevant (min_keep is the floor). "reranker": { "enabled": True, - "type": "ai", + "model_type": "cross-encoder", "strategy": "rerankers-lib", - "model_name": "answerdotai/answerai-colbert-small-v1", - "top_k": 10 + "model_name": EXTERNAL_MODELS["reranker_model"], + "top_k": 10, + "min_score": 0.5, + "min_keep": 3 + }, + # Arm H (2026-08-15): per-sub-query retrieval, pooled + deduped + # candidates, ONE rerank + ONE synthesis over the union context. The + # compose path (answer each sub-question, then compose) remains + # available via compose_from_sub_answers / the UI toggle. + # Two-variant decomposer: single-turn questions use a frozen prompt; + # multi-turn requests get a history-aware variant that resolves + # references to earlier turns. resolve_only skips splitting and uses + # only the resolved query (measured neutral; kept as an option). + "query_decomposition": { + "enabled": True, + "compose_from_sub_answers": False, + "pooled_first_stage": True, + "resolve_only": False }, - "query_decomposition": {"enabled": True, "max_sub_queries": 3}, - "verification": {"enabled": True}, + # Off by default: measured annotate-only (zero verdict flips in ablation). + "verification": {"enabled": False}, "retrieval_k": 20, - "contextual_enricher": {"enabled": True, "window_size": 1} + "context_window_size": 0, + "semantic_cache_threshold": 0.98, + "cache_scope": "session", + "contextual_enricher": {"enabled": True, "window_size": 1}, + "indexing": { + "embedding_batch_size": 50, + "enrichment_batch_size": 10, + "extract_crossrefs": True + } } ``` @@ -444,28 +523,71 @@ LocalGPT offers two main pipeline configurations: "description": "Speed-optimized pipeline with minimal overhead", "retrieval": { "search_type": "vector_only", - "late_chunking": {"enabled": False} + "latechunk": {"enabled": False}, + "dense": {"enabled": True} }, "reranker": {"enabled": False}, "query_decomposition": {"enabled": False}, "verification": {"enabled": False}, "retrieval_k": 10, - "contextual_enricher": {"enabled": False} + "contextual_enricher": {"enabled": False}, + "indexing": { + "embedding_batch_size": 100, + "enrichment_batch_size": 50 + } } ``` -### Search Configuration +### Retrieval Modes + +`retrieval_mode` (wire name; `search_type` inside the pipeline config) accepts: + +| Value | Behaviour | +|-------|-----------| +| `hybrid` *(default)* | Vector and LanceDB full-text legs run in parallel and are fused with Reciprocal Rank Fusion | +| `vector_only` | Dense vector search only | +| `fts_only` | LanceDB full-text search only | + +Anything else is rejected with HTTP 400 by the RAG API. There is no +`dense_weight` / `denseWeight` knob โ€” RRF needs no weights. + +### Experimental: multi-vector (late-interaction) retrieval + +The repo carries env-gated hooks for ColBERT-style multi-vector retrieval, +served by an out-of-process sidecar (SentenceTransformers v6 needs a newer +torch/transformers stack than the pinned in-repo one). All measured, none +default โ€” the full study is in `eval/decisions/multivector-retrieval-2026-08-19.md`, +`paraphrase-robustness-2026-08-20.md` and `union-fusion-2026-08-20.md`: + +| Env | Behaviour | Measured verdict | +|-----|-----------|------------------| +| `MV_RETRIEVAL_ENDPOINT` | Multi-vector MaxSim replaces the dense leg | Loses on both document-phrased and paraphrased queries | +| + `MV_RRF_LEG=1` | Multi-vector runs as a third RRF leg | Break-even; small gain only on paraphrased queries | +| + `MV_UNION=1` | All three legs' candidates are unioned (no RRF cut) and the reranker arbitrates | Best config for paraphrase-heavy / conversational queries (+4/120 real); costs โˆ’3/120 on document-phrased queries | + +Rule of thumb: if your users quote the documents' own vocabulary, keep the +default 2-leg hybrid; if they ask in their own words, `MV_UNION=1` is the +measured winner (at ~30โ€“40% extra query latency plus the sidecar process). + +## ๐Ÿ”ฌ Evaluation + +Every retrieval component in the default profile earned its place in a measured +A/B โ€” and several plausible features are off because they measurably didn't +(late chunking, verification, cross-ref hops, document escalation, multi-vector +retrieval). The harness lives in `eval/`: + +- `eval/goldset/` โ€” five 24-question corpora (technical RFCs, M&A documents, + a service manual, HR policy, this project's docs), a 12-conversation + multi-turn set (`multiturn.jsonl`), and `paraphrases.jsonl` โ€” verified + same-meaning rewrites of all 120 questions with ~0.21 content-word overlap, + for measuring robustness to users who don't phrase queries like the documents. +- `eval/judge.py` โ€” the groundedness judge (deterministic local model or a + stronger LLM via `JUDGE_MODEL`); judged comparisons use blind multi-voter + panels on every changed row. +- `eval/decisions/` โ€” one dated record per experiment: setup, numbers, + flip-level panel verdicts, and the decision. If you want to know why a + default is what it is, the answer is in there. -```python -SEARCH_CONFIG = { - 'hybrid': { - 'dense_weight': 0.7, - 'sparse_weight': 0.3, - 'retrieval_k': 20, - 'reranker_top_k': 10 - } -} -``` --- ## ๐Ÿ› ๏ธ Troubleshooting @@ -475,10 +597,10 @@ SEARCH_CONFIG = { #### Installation Problems ```bash # Check Python version -python --version # Should be 3.8+ +python --version # 3.10+ required, 3.11 recommended # Check dependencies -pip list | grep -E "(torch|transformers|lancedb)" +pip list | grep -E "(torch|transformers|lancedb|docling|rerankers)" # Reinstall dependencies pip install -r requirements.txt --force-reinstall @@ -491,7 +613,8 @@ ollama list curl http://localhost:11434/api/tags # Pull missing models -ollama pull qwen3:0.6b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` #### Database Issues @@ -499,11 +622,17 @@ ollama pull qwen3:0.6b # Check database connectivity python -c "from backend.database import ChatDatabase; db = ChatDatabase(); print('โœ… Database OK')" -# Reset database (WARNING: This deletes all data) +# Reset database (WARNING: This deletes all sessions, messages and index metadata) rm backend/chat_data.db python -c "from backend.database import ChatDatabase; ChatDatabase().init_database()" ``` +#### Dimension mismatch after changing the embedding model +``` +ValueError: ... changing the embedding model requires rebuilding the index +``` +Delete the affected index in the UI (or `DELETE /indexes/{id}`) and rebuild it. + #### Performance Issues ```bash # Check system resources @@ -512,263 +641,241 @@ python system_health_check.py # Monitor memory usage htop # or Task Manager on Windows -# Optimize for low-memory systems -export PYTORCH_CUDA_ALLOC_CONF=max_split_size_mb:512 +# Use lighter models (the default embedder is already the small one at 1.2GB) +export GENERATION_MODEL=qwen3.5:4b ``` ### Getting Help -1. **Check Logs**: The system creates structured logs in the `logs/` directory: - - `logs/system.log`: Main system events and errors - - `logs/ollama.log`: Ollama server logs - - `logs/rag-api.log`: RAG API processing logs - - `logs/backend.log`: Backend server logs - - `logs/frontend.log`: Frontend build and runtime logs +1. **Check Logs**: `run_system.py` writes structured logs to `logs/`: + - `logs/system.log`: launcher events + - `logs/ollama.log`, `logs/rag-api.log`, `logs/backend.log`, `logs/frontend.log`: per-service output + - `logs/run_system.pid`: PIDs used by `--stop` -2. **System Health**: Run comprehensive diagnostics: +2. **System Health**: Run diagnostics: ```bash - python system_health_check.py # Full system diagnostics - python run_system.py --health # Service status check + python system_health_check.py # loads models, runs a sample query + python run_system.py --health # HTTP checks, non-zero exit on failure ``` -3. **Health Endpoints**: Check individual service health: +3. **Health Endpoints**: - Backend: `http://localhost:8000/health` - RAG API: `http://localhost:8001/health` - Ollama: `http://localhost:11434/api/tags` -4. **Documentation**: Check the [Technical Documentation](TECHNICAL_DOCS.md) +4. **Documentation**: See [Documentation/system_overview.md](Documentation/system_overview.md), and [Documentation/design_rationale.md](Documentation/design_rationale.md) for why each component is built the way it is โ€” with the evidence and the eval numbers behind every default, plus a "deliberately not implemented" list 5. **GitHub Issues**: Report bugs and request features -6. **Community**: Join our Discord/Slack community +6. **Community**: Join our Discord --- ## ๐Ÿ”— API Reference -### Core Endpoints +Two HTTP services. The backend gateway on **:8000** owns sessions, indexes, +uploads and chat history; the RAG API on **:8001** owns retrieval and indexing. +Both accept `snake_case` and `camelCase` spellings of every option and normalise +them to one canonical key. + +### Backend gateway โ€” http://localhost:8000 + +```http +GET /health # {status, ollama_running, available_models, database_stats} +GET /models # {generation_models, embedding_models} + +GET /sessions # {sessions, total} +POST /sessions # {title?, model?} -> 201 {session, session_id} +GET /sessions/{id} # {session, messages} +DELETE /sessions/{id} # {deleted: true} +GET /sessions/cleanup # removes empty sessions +POST /sessions/{id}/rename # {title} -> {message, session} +GET /sessions/{id}/documents # {session, files, file_count} +GET /sessions/{id}/indexes # {indexes, total} +POST /sessions/{id}/indexes/{index_id} # link an index to a session +POST /sessions/{id}/upload # multipart/form-data, field "files" +POST /sessions/{id}/index # index this session's uploads +POST /sessions/{id}/messages # chat (see below) + +GET /indexes # {indexes, total} +POST /indexes # {name, description?, metadata?} -> 201 {index_id} +GET /indexes/{id} +DELETE /indexes/{id} # also drops the LanceDB table +POST /indexes/{id}/upload # multipart/form-data, field "files" +POST /indexes/{id}/build # build/rebuild from uploaded documents + +POST /chat # session-less Ollama chat, no retrieval +``` + +#### Session chat -#### Chat API ```http -# Session-based chat (recommended) -POST /sessions/{session_id}/chat +POST /sessions/{session_id}/messages Content-Type: application/json { - "query": "What are the main topics discussed?", - "search_type": "hybrid", + "message": "What are the main topics discussed?", + "model": "qwen3.5:9b", + "retrieval_mode": "hybrid", "retrieval_k": 20, + "reranker_top_k": 10, + "context_window_size": 1, "ai_rerank": true, - "context_window_size": 5 -} - -# Legacy chat endpoint -POST /chat -Content-Type: application/json - -{ - "query": "What are the main topics discussed?", - "session_id": "uuid", - "search_type": "hybrid", - "retrieval_k": 20 + "context_expand": true, + "query_decompose": true, + "compose_sub_answers": true, + "verify": true, + "provence_prune": false, + "provence_threshold": 0.1, + "force_rag": false } ``` -#### Index Management -```http -# Create index -POST /indexes -Content-Type: application/json -{ - "name": "My Index", - "description": "Description", - "config": "default" -} - -# Get all indexes -GET /indexes - -# Get specific index -GET /indexes/{id} - -# Upload documents to index -POST /indexes/{id}/upload -Content-Type: multipart/form-data -files: [file1.pdf, file2.pdf, ...] - -# Build index (process uploaded documents) -POST /indexes/{id}/build -Content-Type: application/json +Response: +```json { - "config_mode": "default", - "enable_enrich": true, - "chunk_size": 512 + "response": "โ€ฆ", + "session": { "...": "updated session row" }, + "source_documents": [], + "used_rag": true } - -# Delete index -DELETE /indexes/{id} ``` -#### Session Management -```http -# Create session -POST /sessions -Content-Type: application/json -{ - "title": "My Session", - "model": "qwen3:0.6b" -} - -# Get all sessions -GET /sessions - -# Get specific session -GET /sessions/{session_id} - -# Get session documents -GET /sessions/{session_id}/documents +The backend decides per message whether to answer directly with Ollama or to +forward to the RAG API. `force_rag: true` skips that decision and always calls the +RAG API. Both the user message and the answer are written to SQLite on this path. -# Get session indexes -GET /sessions/{session_id}/indexes +### RAG API โ€” http://localhost:8001 -# Link index to session -POST /sessions/{session_id}/indexes/{index_id} - -# Delete session -DELETE /sessions/{session_id} - -# Rename session -POST /sessions/{session_id}/rename -Content-Type: application/json -{ - "new_title": "Updated Session Name" -} +```http +GET /health # {"status": "ok"} +GET /models # {generation_models, embedding_models} +POST /chat # {answer, source_documents} +POST /chat/stream # Server-Sent Events, terminated by a "complete" event +POST /index # run the indexing pipeline over file_paths ``` -### Advanced Features - -#### Query Decomposition -The system can break complex queries into sub-questions for better answers: -```http -POST /sessions/{session_id}/chat -Content-Type: application/json +#### `POST /chat` and `POST /chat/stream` +```json { - "query": "Compare the methodologies and analyze their effectiveness", + "query": "Explain the methodology", + "session_id": "uuid", + "table_name": "text_pages_", + "model": "qwen3.5:9b", + "retrieval_mode": "hybrid", + "retrieval_k": 20, + "context_window_size": 1, + "reranker_top_k": 10, + "ai_rerank": true, + "context_expand": true, "query_decompose": true, - "compose_sub_answers": true + "compose_sub_answers": true, + "verify": true, + "force_rag": false, + "provence_prune": false, + "provence_threshold": 0.1 } ``` -#### Answer Verification -Independent verification pass for accuracy using a separate verification model: -```http -POST /sessions/{session_id}/chat -Content-Type: application/json +`/chat` returns `{"answer": "...", "source_documents": [...]}`. There is no +top-level `confidence` field โ€” when verification runs it appends +`[Confidence: N%]` (and a low-confidence warning) to the answer text itself. -{ - "query": "What are the key findings?", - "verify": true -} -``` - -#### Contextual Enrichment -Document context enrichment during indexing for better understanding: -```bash -# Enable during index building -POST /indexes/{id}/build -{ - "enable_enrich": true, - "window_size": 2 -} -``` +`/chat/stream` emits `data: {"type": "", "data": {...}}` lines and ends +with a `complete` event carrying the same object `/chat` would return. -#### Late Chunking -Better context preservation by chunking after embedding: -```bash -# Configure in pipeline -"late_chunking": {"enabled": true} -``` +`force_rag: true` skips the agent's triage step so the query always goes through +retrieval; `verify`, `ai_rerank`, `query_decompose` and `compose_sub_answers` +still apply. An unsupported `retrieval_mode` is rejected with HTTP 400. -#### Streaming Chat -```http -POST /chat/stream -Content-Type: application/json +#### `POST /index` +```json { - "query": "Explain the methodology", + "file_paths": ["/abs/path/doc1.pdf", "/abs/path/doc2.pdf"], "session_id": "uuid", - "stream": true + "table_name": "text_pages_", + "chunk_size": 512, + "window_size": 2, + "retrieval_mode": "hybrid", + "enable_enrich": true, + "enable_latechunk": false, + "enable_docling_chunk": true, + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", + "batch_size_embed": 50, + "batch_size_enrich": 25 } ``` -#### Batch Processing -```bash -# Using the batch indexing script -python demo_batch_indexing.py --config batch_indexing_config.json +`file_paths` is required; the values above are the defaults applied when a field +is omitted. Response: -# Example batch configuration (batch_indexing_config.json): +```json { - "index_name": "Sample Batch Index", - "index_description": "Example batch index configuration", - "documents": [ - "./rag_system/documents/invoice_1039.pdf", - "./rag_system/documents/invoice_1041.pdf" - ], - "processing": { + "message": "Indexing process for 2 file(s) completed successfully.", + "table_name": "text_pages_", + "latechunk": false, + "docling_chunk": true, + "indexing_config": { "chunk_size": 512, - "chunk_overlap": 64, - "enable_enrich": true, - "enable_latechunk": true, - "enable_docling": true, - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", - "generation_model": "qwen3:0.6b", "retrieval_mode": "hybrid", - "window_size": 2 + "window_size": 2, + "enable_enrich": true, + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", + "batch_size_embed": 50, + "batch_size_enrich": 25 } } ``` -```http -# API endpoint for batch processing -POST /batch/index -Content-Type: application/json +`retrieval_mode` at index time is validated and recorded with the index config; +it takes effect at query time. `enable_docling_chunk` defaults to `true` +(Docling structure-aware chunking); sending `false` selects the legacy chunker. +Indexing is synchronous โ€” the call returns when +the pipeline finishes, which is why the backend allows up to +`RAG_API_INDEX_TIMEOUT` (default 3600s) for it. -{ - "file_paths": ["doc1.pdf", "doc2.pdf"], - "config": { - "chunk_size": 512, - "enable_enrich": true, - "enable_latechunk": true, - "enable_docling": true - } -} -``` +For the full route table see [Documentation/api_reference.md](Documentation/api_reference.md). -For complete API documentation, see [API_REFERENCE.md](API_REFERENCE.md). +### Known limitations + +- **The RAG API is single-threaded.** Requests are serialised: one chat or + indexing run at a time. The backend gateway is threaded, so it stays responsive, + but a long RAG call blocks the next one. +- **Streamed turns are persisted after the fact.** The chat UI streams from + `:8001/chat/stream` directly and, when the stream completes, saves the finished + turn through the gateway (`POST /sessions/{id}/messages/save`). If the browser is + closed mid-stream, that turn is not saved. +- **Index metadata is per index, not per session.** Choosing a different embedding + model requires rebuilding the index. --- ## ๐Ÿ—๏ธ Architecture -LocalGPT is built with a modular, scalable architecture: - ```mermaid graph TB - UI[Web Interface] --> API[Backend API] - API --> Agent[RAG Agent] + UI[Next.js UI :3000] --> API[Backend gateway :8000] + UI -. "SSE /chat/stream" .-> RAGAPI + API --> RAGAPI[RAG API :8001] + RAGAPI --> Agent[RAG Agent] Agent --> Retrieval[Retrieval Pipeline] - Agent --> Generation[Generation Pipeline] + Agent --> Ollama[Ollama :11434] - Retrieval --> Vector[Vector Search] - Retrieval --> BM25[BM25 Search] - Retrieval --> Rerank[Reranking] + Retrieval --> Vector[Vector search] + Retrieval --> FTS[LanceDB full-text search] + Vector --> RRF[Reciprocal Rank Fusion] + FTS --> RRF + RRF --> Rerank["Cross-encoder rerank (on by default)"] Vector --> LanceDB[(LanceDB)] - BM25 --> BM25DB[(BM25 Index)] + FTS --> LanceDB - Generation --> Ollama[Ollama Models] - Generation --> HF[Hugging Face Models] - - API --> SQLite[(SQLite DB)] + API --> SQLite[(SQLite: sessions, messages, indexes)] + RAGAPI --> SQLite ``` Overview of the Retrieval Agent @@ -778,53 +885,53 @@ graph TD classDef llmcall fill:#e6f3ff,stroke:#007bff; classDef pipeline fill:#e6ffe6,stroke:#28a745; classDef cache fill:#fff3e0,stroke:#fd7e14; - classDef logic fill:#f8f9fa,stroke:#6c757d; - classDef thread stroke-dasharray: 5 5; - A(Start: Agent.run) --> B_asyncio.run(_run_async); - B --> C{_run_async}; + A(Start: Agent.run) --> C{_run_async}; - C --> C1[Get Chat History]; - C1 --> T1[Build Triage Prompt
Query + Doc Overviews ]; - T1 --> T2["(asyncio.to_thread)
LLM Triage: RAG or LLM_DIRECT?"]; class T2 llmcall,thread; + C --> C1[Get chat history]; + C1 --> T0{force_rag?}; + T0 -- Yes --> RAG_Path; + T0 -- No --> T1[Route via document overviews]; + T1 --> T2["LLM triage fallback:
rag_query | direct_answer"]; class T2 llmcall; T2 --> T3{Decision?}; - T3 -- RAG --> RAG_Path; - T3 -- LLM_DIRECT --> LLM_Path; + T3 -- rag_query --> RAG_Path; + T3 -- direct_answer --> LLM_Path; subgraph RAG Path - RAG_Path --> R1[Format Query + History]; - R1 --> R2["(asyncio.to_thread)
Generate Query Embedding"]; class R2 pipeline,thread; - R2 --> R3{{Check Semantic Cache}}; class R3 cache; - R3 -- Hit --> R_Cache_Hit(Return Cached Result); - R_Cache_Hit --> R_Hist_Update; - R3 -- Miss --> R4{Decomposition
Enabled?}; - - R4 -- Yes --> R5["(asyncio.to_thread)
Decompose Raw Query"]; class R5 llmcall,thread; - R5 --> R6{{Run Sub-Queries
Parallel RAG Pipeline}}; class R6 pipeline,thread; - R6 --> R7[Collect Results & Docs]; - R7 --> R8["(asyncio.to_thread)
Compose Final Answer"]; class R8 llmcall,thread; - R8 --> V1(RAG Answer); - - R4 -- No --> R9["(asyncio.to_thread)
Run Single Query
(RAG Pipeline)"]; class R9 pipeline,thread; + RAG_Path --> R1[Format query + history]; + R1 --> R2[Embed query]; class R2 pipeline; + R2 --> R3{{Semantic cache
threshold 0.98, session-scoped}}; class R3 cache; + R3 -- Hit --> FinalResult; + R3 -- Miss --> R4{Decomposition enabled?}; + + R4 -- Yes --> R5[Decompose query]; class R5 llmcall; + R5 --> R6{{Retrieve per sub-query, then pool + dedupe the candidates}}; class R6 pipeline; + R6 --> R8[One rerank + one synthesis over the pooled context]; class R8 llmcall; + R8 --> V1(RAG answer); + + R4 -- No --> R9[Run single query through the retrieval pipeline]; class R9 pipeline; R9 --> V1; - V1 --> V2{{Verification
await verify_async}}; class V2 llmcall; - V2 --> V3(Final RAG Result); - V3 --> R_Cache_Store{{Store in Semantic Cache}}; class R_Cache_Store cache; + V1 --> V2{{Verification}}; class V2 llmcall; + V2 --> R_Cache_Store{{Store in semantic cache}}; class R_Cache_Store cache; R_Cache_Store --> FinalResult; end subgraph Direct LLM Path - LLM_Path --> L1[Format Query + History]; - L1 --> L2["(asyncio.to_thread)
Generate Direct LLM Answer
(No RAG)"]; class L2 llmcall,thread; - L2 --> FinalResult(Final Direct Result); + LLM_Path --> L2[Generate answer without retrieval]; class L2 llmcall; + L2 --> FinalResult(Final result); end - FinalResult --> R_Hist_Update(Update Chat History); - R_Hist_Update --> ZZZ(End: Return Result); + FinalResult --> R_Hist_Update(Update in-memory chat history); + R_Hist_Update --> ZZZ["End: return answer + source_documents"]; ``` +Inside the retrieval pipeline a query runs: embed โ†’ hybrid retrieve (vector + +FTS, fused with RRF) โ†’ optional late-chunk leg โ†’ cross-encoder rerank (on by +default) โ†’ context window expansion โ†’ optional Provence sentence +pruning โ†’ synthesis. + --- ## ๐Ÿค Contributing @@ -844,7 +951,8 @@ npm install # Install Ollama and models curl -fsSL https://ollama.ai/install.sh | sh -ollama pull qwen3:0.6b qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b # Verify setup python system_health_check.py @@ -879,7 +987,7 @@ This project is licensed under the MIT License - see the [LICENSE](LICENSE) file ## ๐Ÿ“ž Support -- **Documentation**: [Technical Docs](TECHNICAL_DOCS.md) +- **Documentation**: [Documentation/system_overview.md](Documentation/system_overview.md) - **Issues**: [GitHub Issues](https://github.com/PromtEngineer/localGPT/issues) - **Discussions**: [GitHub Discussions](https://github.com/PromtEngineer/localGPT/discussions) - **Business Deployment and Customization**: [Contact Us](https://tally.so/r/wv6R2d) @@ -890,3 +998,5 @@ This project is licensed under the MIT License - see the [LICENSE](LICENSE) file ## Star History [![Star History Chart](https://api.star-history.com/svg?repos=PromtEngineer/localGPT&type=Date)](https://star-history.com/#PromtEngineer/localGPT&Date) + + diff --git a/WATSONX_README.md b/WATSONX_README.md index a21bcbc6..76c9f993 100644 --- a/WATSONX_README.md +++ b/WATSONX_README.md @@ -1,91 +1,126 @@ # Watson X Integration with Granite Models -This branch adds support for IBM Watson X AI with Granite models as an alternative to Ollama for running LocalGPT. +localGPT can run its LLM calls against IBM watsonx.ai Granite models instead of a local +Ollama server. ## Overview -LocalGPT now supports two LLM backends: -1. **Ollama** (default): Run models locally using Ollama -2. **Watson X**: Use IBM's Granite models hosted on Watson X AI +`rag_system` supports two LLM backends, selected with the `LLM_BACKEND` environment +variable: -## What Changed +1. **Ollama** (`ollama`, the default) โ€” models run locally. +2. **Watson X** (`watsonx`) โ€” Granite models hosted on IBM watsonx.ai. -- Added `WatsonXClient` class in `rag_system/utils/watsonx_client.py` that provides an Ollama-compatible interface for Watson X -- Updated `factory.py` and `main.py` to support backend switching via environment variable -- Added `ibm-watsonx-ai` SDK dependency to `requirements.txt` -- Configuration now supports both backends through environment variables +The switch is made in `rag_system/factory.py::_build_llm_client()`, which returns either an +`OllamaClient` or a `WatsonXClient` (`rag_system/utils/watsonx_client.py`) together with +the matching config dict (`OLLAMA_CONFIG` or `WATSONX_CONFIG` from `rag_system/main.py`). +Both dicts expose the same `generation_model` / `enrichment_model` keys, so the agent, the +retrieval pipeline and the indexing pipeline are unchanged. -## Prerequisites +### What the backend switch does and does not cover + +Switched to Watson X: + +- Answer generation and sub-answer composition (`generation_model`). +- Query routing, triage, query decomposition, contextual enrichment, document overviews and + answer verification (`enrichment_model`). -To use Watson X with Granite models, you need: +**Always local, regardless of `LLM_BACKEND`:** + +- **Embeddings.** `rag_system/indexing/representations.py::select_embedder()` returns a + Hugging Face model when `EMBEDDING_MODEL` contains a `/`, and an Ollama embedder + otherwise. There is no Watson X embedding path, and `WatsonXClient` exposes no embedding + method. Point `EMBEDDING_MODEL` at a Hugging Face repo (the default + `microsoft/harrier-oss-v1-0.6b`) so no Ollama server is needed for indexing. +- **Reranking** (on by default; `Qwen/Qwen3-Reranker-4B`, loaded lazily on the first + reranked query) and **Provence + sentence pruning** โ€” both are local `transformers` models. +- **The backend gateway's direct-LLM path.** `backend/server.py` answers non-document + questions through `backend/ollama_client.py`, which always + talks to `OLLAMA_HOST`. The routing decision itself is the deterministic, no-LLM + `should_use_rag` gate. Watson X only serves requests that reach the RAG API on port + 8001. + +## Prerequisites -1. IBM Cloud account with Watson X access -2. Watson X API key -3. Watson X project ID +1. IBM Cloud account with watsonx.ai access +2. A watsonx.ai API key +3. A watsonx.ai project ID -### Getting Your Credentials +### Getting your credentials 1. Go to [IBM Cloud](https://cloud.ibm.com/) -2. Navigate to Watson X AI service +2. Navigate to the watsonx.ai service 3. Create or select a project 4. Get your API key from IBM Cloud IAM -5. Copy your project ID from the Watson X project settings +5. Copy your project ID from the project settings + +## Installation + +The SDK is **not** installed by the root `requirements.txt` (it is listed there as a +commented-out optional extra). Install it explicitly: + +```bash +pip install "ibm-watsonx-ai>=1.3.39" +``` + +`rag_system/requirements.txt` โ€” the RAG-only dependency list โ€” does pin it, so +`pip install -r rag_system/requirements.txt` also gets you the SDK. + +Without the package, `WatsonXClient.__init__` raises +`ImportError: ibm-watsonx-ai package is required.` as soon as the agent is constructed. ## Configuration -### Environment Variables +Copy the example file and fill in your credentials: + +```bash +cp env.example.watsonx .env +``` -Create a `.env` file or set these environment variables: +The variables, with the defaults from `rag_system/main.py`: ```bash # Choose LLM backend (default: ollama) LLM_BACKEND=watsonx -# Watson X Configuration +# Watson X credentials WATSONX_API_KEY=your_api_key_here WATSONX_PROJECT_ID=your_project_id_here WATSONX_URL=https://us-south.ml.cloud.ibm.com -# Model Configuration +# Model configuration WATSONX_GENERATION_MODEL=ibm/granite-13b-chat-v2 WATSONX_ENRICHMENT_MODEL=ibm/granite-8b-japanese ``` -### Available Granite Models +`WATSONX_API_KEY` and `WATSONX_PROJECT_ID` are mandatory: `_build_llm_client()` raises +`ValueError: Watson X configuration incomplete.` when either is empty. -Watson X offers several Granite models: -- `ibm/granite-13b-chat-v2` - General purpose chat model -- `ibm/granite-13b-instruct-v2` - Instruction-following model -- `ibm/granite-20b-multilingual` - Multilingual support -- `ibm/granite-8b-japanese` - Lightweight Japanese model -- `ibm/granite-3b-code-instruct` - Code generation model +Use model ids that exist in your watsonx.ai instance โ€” the two above are only the code +defaults, and `ibm/granite-8b-japanese` in particular is unlikely to be the utility model +you want. IBM's +[supported foundation models](https://www.ibm.com/docs/en/watsonx/saas?topic=solutions-supported-foundation-models) +page lists what is currently available. -For a full list of available models, visit the [Watson X documentation](https://www.ibm.com/docs/en/watsonx/saas?topic=solutions-supported-foundation-models). - -## Installation +## Usage -1. Install the Watson X SDK: -```bash -pip install ibm-watsonx-ai>=1.3.39 -``` +### Running with Watson X -Or install all dependencies: ```bash -pip install -r rag_system/requirements.txt +export LLM_BACKEND=watsonx +python -m rag_system.main api # RAG API on port 8001 ``` -## Usage - -### Running with Watson X - -Once configured, simply set the environment variable and run as normal: +`python -m rag_system.api_server` is equivalent. To index or ask a one-off question from +the CLI: ```bash -export LLM_BACKEND=watsonx -python -m rag_system.main api +python -m rag_system.main index ./shared_uploads +python -m rag_system.main chat "What is in these documents?" ``` -Or in Python: +Or programmatically: ```python import os @@ -93,133 +128,152 @@ os.environ['LLM_BACKEND'] = 'watsonx' from rag_system.factory import get_agent -# Get agent with Watson X backend agent = get_agent(mode="default") - -# Use as normal result = agent.run("What is artificial intelligence?") -print(result) +print(result["answer"]) ``` -### Switching Between Backends - -You can easily switch between Ollama and Watson X: +### Switching between backends ```bash -# Use Ollama (local) +# Local Ollama export LLM_BACKEND=ollama python -m rag_system.main api -# Use Watson X (cloud) +# Watson X export LLM_BACKEND=watsonx python -m rag_system.main api ``` -## Features +### Using Watson X from the web UI + +Document-grounded answers do come from Watson X, with two caveats. -The Watson X client supports all the key features used by LocalGPT: +**The model dropdown still lists Ollama tags.** The UI populates it from the *backend +gateway's* `GET :8000/models` (`src/lib/api.ts`), which always queries Ollama. Only the +RAG API's own `GET :8001/models` is backend-aware and returns the configured Granite ids +when `LLM_BACKEND=watsonx`. -- โœ… Text generation / completion -- โœ… Async generation -- โœ… Streaming responses -- โœ… Embeddings (if using Watson X embedding models) -- โœ… Custom generation parameters (temperature, max_tokens, top_p, top_k) -- โš ๏ธ Image/multimodal support (limited, depends on model availability) +**That mismatch is harmless.** A per-request `model` is applied only when it is valid for +the active backend โ€” under Watson X the id must contain a `/`, so a stray Ollama tag such +as `qwen3.5:9b` is ignored with a warning and `WATSONX_GENERATION_MODEL` is used instead. +The override is also scoped to that single request and restored afterwards, so one user's +choice cannot leak into the next request. -## API Compatibility +The backend gateway also keeps using Ollama for its non-document fast path (its routing +decision is the deterministic, no-LLM `should_use_rag` gate โ€” see above), so a fully +Ollama-free setup means talking to the RAG API on +port 8001 directly. -The `WatsonXClient` provides the same interface as `OllamaClient`: +## API compatibility + +`WatsonXClient` implements the three methods the RAG system uses from `OllamaClient`: ```python from rag_system.utils.watsonx_client import WatsonXClient client = WatsonXClient( api_key="your_api_key", - project_id="your_project_id" + project_id="your_project_id", ) -# Generate completion +# Blocking completion -> {"response": str, "model": str, "done": True} response = client.generate_completion( model="ibm/granite-13b-chat-v2", - prompt="Explain quantum computing" + prompt="Explain quantum computing", ) - print(response['response']) -# Stream completion +# Streaming completion -> yields text chunks for chunk in client.stream_completion( model="ibm/granite-13b-chat-v2", - prompt="Write a story about AI" + prompt="Write a story about AI", ): print(chunk, end='', flush=True) ``` +There is also `generate_completion_async()`, which runs the blocking call in an executor +(the IBM SDK has no native async API). The verifier uses it. + ## Limitations -1. **Embedding Models**: Watson X uses different embedding models than Ollama. Make sure to configure embedding models appropriately in `main.py` if needed. +1. **No JSON mode.** `OllamaClient.generate_completion()` forwards `format="json"` to + Ollama; `WatsonXClient.generate_completion()` accepts the argument and ignores it. + Triage, the overview router, query decomposition and the verifier all rely on the model + returning parseable JSON. When parsing fails the code falls back to safe defaults + (route to RAG, do not decompose, no confidence tag), so answers still work but routing + and verification are less reliable than on Ollama. -2. **Multimodal Support**: Image support varies by model availability in Watson X. Not all Granite models support multimodal inputs. +2. **Embeddings and rerankers are never Watson X.** See the section above โ€” they load from + Hugging Face (or Ollama) in-process. -3. **Streaming**: Streaming support depends on the Watson X SDK version and may fall back to returning the full response at once. +3. **Streaming.** `stream_completion()` uses the SDK's `generate_text_stream()` and falls + back to yielding the full response as a single chunk if that method is unavailable. -4. **Rate Limits**: Watson X has API rate limits that may differ from local Ollama usage. Monitor your usage accordingly. +4. **Errors are swallowed into empty responses.** `generate_completion()` catches + exceptions and returns `{"response": "", "error": ...}`, so an authentication or quota + failure shows up as an empty answer rather than an HTTP error. Check the server log. + +5. **Rate limits and cost.** Watson X is a metered cloud service; local Ollama is not. ## Troubleshooting -### Authentication Errors +### `ImportError: ibm-watsonx-ai package is required` + +Run `pip install "ibm-watsonx-ai>=1.3.39"` in the environment that runs the RAG API. -If you see authentication errors: -- Verify your API key is correct -- Check that your project ID matches an existing Watson X project -- Ensure your IBM Cloud account has Watson X access +### `ValueError: Watson X configuration incomplete` -### Model Not Found +`WATSONX_API_KEY` or `WATSONX_PROJECT_ID` is empty. If you put them in `.env`, make sure the +process is started from the repository root โ€” `load_dotenv()` resolves the file relative to +the working directory. -If you get model not found errors: -- Verify the model ID is correct (e.g., `ibm/granite-13b-chat-v2`) -- Check that the model is available in your Watson X instance -- Some models may require additional permissions +### Authentication errors -### Connection Errors +- Verify the API key is correct +- Check that the project ID matches an existing watsonx.ai project +- Ensure your IBM Cloud account has watsonx.ai access + +### Model not found + +- Verify the model id (e.g. `ibm/granite-13b-chat-v2`) +- Check that the model is available in your instance and region +- Some models require additional entitlements + +### Connection errors -If you experience connection issues: - Check your internet connection -- Verify the Watson X URL is correct for your region -- Check IBM Cloud status page for service outages +- Verify `WATSONX_URL` matches your region +- Check the IBM Cloud status page -## Cost Considerations +### Empty answers with no visible error -Unlike local Ollama, Watson X is a cloud service with usage-based pricing: -- Token-based pricing for generation -- Consider your query volume -- Monitor usage through IBM Cloud dashboard +See limitation 4 โ€” look at the RAG API's stdout for `Error generating completion: โ€ฆ`. ## Reverting to Ollama -To switch back to local Ollama: - ```bash -unset LLM_BACKEND # or set LLM_BACKEND=ollama +unset LLM_BACKEND # or set LLM_BACKEND=ollama python -m rag_system.main api ``` +Indexes built while Watson X was active remain valid: the embeddings were produced locally +and never touched Watson X. + ## Support -For Watson X specific issues: -- [IBM Watson X Documentation](https://www.ibm.com/docs/en/watsonx/saas) -- [Watson X Developer Hub](https://www.ibm.com/watsonx/developer/) +For Watson X issues: +- [IBM watsonx Documentation](https://www.ibm.com/docs/en/watsonx/saas) +- [watsonx Developer Hub](https://www.ibm.com/watsonx/developer/) - [IBM Cloud Support](https://cloud.ibm.com/docs/get-support) -For LocalGPT issues: -- [LocalGPT GitHub Issues](https://github.com/PromtEngineer/localGPT/issues) +For localGPT issues: +- [localGPT GitHub Issues](https://github.com/PromtEngineer/localGPT/issues) ## Contributing -If you find issues with the Watson X integration or want to add features: -1. Create an issue describing the problem/feature -2. Submit a pull request with your changes -3. Ensure all tests pass +See [CONTRIBUTING.md](CONTRIBUTING.md). ## License -This integration follows the same license as LocalGPT (MIT License). +This integration follows the same license as localGPT (MIT License). diff --git a/backend/README.md b/backend/README.md index 0a74e565..baa32bdd 100644 --- a/backend/README.md +++ b/backend/README.md @@ -1,93 +1,185 @@ # localGPT Backend -Simple Python backend that connects your frontend to Ollama for local LLM chat. +The gateway between the Next.js frontend and the rest of the system. It owns chat sessions, +uploaded documents and index bookkeeping in SQLite, answers simple queries directly with +Ollama, and forwards document-grounded queries and indexing jobs to the RAG API. + +``` +frontend :3000 โ”€โ”€โ–บ backend/server.py :8000 โ”€โ”€โ–บ rag_system/api_server.py :8001 โ”€โ”€โ–บ Ollama :11434 + โ”‚ + โ””โ”€โ”€โ–บ Ollama :11434 (direct, non-RAG answers) +``` + +Built on the standard library's `http.server` (`ThreadingTCPServer`, so requests are +handled concurrently). No web framework. ## Prerequisites -1. **Install Ollama** (if not already installed): - ```bash - # Visit https://ollama.ai or run: - curl -fsSL https://ollama.ai/install.sh | sh - ``` +1. **Python 3.10+** (3.11 recommended). -2. **Start Ollama**: +2. **Ollama** running locally: ```bash + # https://ollama.com/download, or: + curl -fsSL https://ollama.ai/install.sh | sh ollama serve ``` -3. **Pull a model** (optional, server will suggest if needed): +3. **Models** โ€” the defaults the backend resolves to: ```bash - ollama pull llama3.2 + ollama pull qwen3.5:9b # generation + ollama pull qwen3.5:4b # enrichment / utility (used by the RAG API, not by the gateway) ``` +4. **The RAG API** (`python -m rag_system.api_server`) if you want document-grounded + answers or indexing. Without it, `/sessions//messages` still works for non-document + queries, and RAG queries return a "could not connect" message. + ## Setup -1. **Install Python dependencies**: - ```bash - pip install -r requirements.txt - ``` +```bash +# From the repository root +pip install -r backend/requirements.txt # requests, python-dotenv -2. **Test Ollama connection**: - ```bash - python ollama_client.py - ``` +# Smoke-test the Ollama connection (note: this pulls GENERATION_MODEL if it is missing) +python backend/ollama_client.py -3. **Start the backend server**: - ```bash - python server.py - ``` +# Start the server +python backend/server.py +``` -Server will run on `http://localhost:8000` +The server listens on `http://localhost:8000`. Run it from the repository root so the +relative paths it uses (`shared_uploads/`, `index_store/`, `lancedb/`, `backend/chat_data.db`) +resolve the same way they do in the Docker image. -## API Endpoints +Usually you do not start it by hand โ€” `python run_system.py` launches Ollama, the RAG API, +the backend and the frontend together. -### Health Check -```bash -GET /health -``` -Returns server status and available models. +## Configuration -### Chat -```bash -POST /chat -Content-Type: application/json - -{ - "message": "Hello!", - "model": "llama3.2:latest", - "conversation_history": [] -} -``` +All environment variables are optional; the value shown is the code default. -Returns: -```json -{ - "response": "Hello! How can I help you?", - "model": "llama3.2:latest", - "message_count": 1 -} -``` +| Variable | Default | Used for | +| --- | --- | --- | +| `OLLAMA_HOST` | `http://localhost:11434` | Direct Ollama calls (`backend/ollama_client.py`). | +| `RAG_API_URL` | `http://localhost:8001` | Base URL for the RAG API; `/chat` and `/index` are built from it. | +| `RAG_API_TIMEOUT` | `600` | Seconds to wait for a RAG chat response before returning 504. | +| `RAG_API_INDEX_TIMEOUT` | `3600` | Seconds to wait for a RAG indexing run before returning 504. | +| `GENERATION_MODEL` | `qwen3.5:9b` | Default answer model. Resolved as env var โ†’ `rag_system.main.OLLAMA_CONFIG` โ†’ literal default. | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | Recorded in index metadata as the enrich/overview model. The gateway no longer calls it (routing is local since Phase 2.3). | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | Recorded in the default index metadata. | +| `DB_PATH` | `backend/chat_data.db` (`/app/backend/chat_data.db` in Docker) | SQLite file for sessions, messages, indexes and documents. | +| `LANCEDB_PATH` | `storage.lancedb_uri` from the default pipeline profile (`./lancedb`) | Vector store the backend drops tables from when an index is deleted. | + +The backend imports `PIPELINE_CONFIGS` and `OLLAMA_CONFIG` from `rag_system.main` when the +package is importable, and degrades gracefully (printing a warning) when it is not. + +## Request routing + +`POST /sessions//messages` decides per message whether to use RAG. The decision is +made by `should_use_rag(message, idx_ids, force_rag)` โ€” a module-level function in +`server.py` with no LLM call, no file reads and no network I/O: + +1. `force_rag` (or `forceRag`) in the body โ†’ RAG, unconditionally. +2. No indexes linked to the session โ†’ direct Ollama answer. +3. The whole message is smalltalk (`hello`, `thanks!`, `bye`, `ok` โ€” a whole-message + allowlist regex, max six words) or a question about the assistant itself + (`who are you`, `what model are you`) โ†’ direct Ollama answer. +4. Everything else โ†’ RAG. + +RAG queries are forwarded to `POST {RAG_API_URL}/chat`; the answer and `source_documents` +come back from there. Direct answers call Ollama with thinking disabled. + +The gate is intentionally biased toward RAG ("escalate, don't pre-decide"). The RAG API's +agent triage runs on every forwarded request and can still answer directly, so the gateway +only has to keep greetings out of the retrieval pipeline โ€” it does not have to be right +about which questions the documents can answer. The pre-retrieval enrichment-model router +and its keyword/length fallback (which misrouted any question containing the word "test") +were removed in Phase 2.3; `Documentation/research/` has the evidence. + +`backend/test_gateway_routing.py` covers the gate: `.venv/bin/python backend/test_gateway_routing.py`. + +**Option casing.** Chat and index options are accepted in both `camelCase` and +`snake_case` and normalised to one canonical `snake_case` key before being forwarded +(`rerankerTopK` โ†’ `reranker_top_k`, `enableLatechunk` โ†’ `enable_latechunk`, and so on). +Options the caller omits are left out of the payload entirely so the RAG pipeline's own +defaults apply. + +**Chat history.** This server is the only writer of chat message rows. The frontend's +streaming path talks to `:8001/chat/stream` directly and then persists the completed +turn here via `POST /sessions//messages/save`. + +## API Endpoints + +`Documentation/api_reference.md` has the request/response detail; this is the route table. + +### GET + +| Route | Returns | +| --- | --- | +| `/health` | `{status, ollama_running, available_models, database_stats}` | +| `/sessions` | `{sessions, total}` | +| `/sessions/cleanup` | `{message, cleanup_count}` โ€” deletes sessions with no messages | +| `/sessions/` | `{session, messages}` | +| `/sessions//documents` | `{session, files, file_count}` | +| `/sessions//indexes` | `{indexes, total}` | +| `/models` | `{generation_models, embedding_models}` | +| `/indexes` | `{indexes, total}` | +| `/indexes/` | the index record, or 404 | + +### POST + +| Route | Body | Returns | +| --- | --- | --- | +| `/chat` | `{message, model?, conversation_history?}` | `{response, model, message_count}` โ€” legacy sessionless path, always direct Ollama | +| `/sessions` | `{title?, model?}` | `{session, session_id}` (201) | +| `/sessions//messages` | `{message, model?, force_rag?, โ€ฆchat options}` | `{response, session, source_documents, used_rag}` | +| `/sessions//upload` | `multipart/form-data`, field `files` | `{message, uploaded_files}` | +| `/sessions//index` | optional JSON index options | the RAG API's `/index` response, or `{message: "No documents to index for this session."}` | +| `/sessions//rename` | `{title}` | `{message, session}` | +| `/sessions//indexes/` | โ€” | `{message}` โ€” links an index to a session | +| `/indexes` | `{name, description?, metadata?}` | `{index_id}` (201) | +| `/indexes//upload` | `multipart/form-data`, field `files` | `{message, uploaded_files}` | +| `/indexes//build` | optional JSON index options | `{response, โ€ฆapplied options}` | + +### DELETE + +| Route | Returns | +| --- | --- | +| `/sessions/` | `{deleted: true}`, or 404 | +| `/indexes/` | `{message, index_id}`, or 404 โ€” also drops the index's LanceDB table | + +`OPTIONS` on any path returns the CORS preflight headers (`GET, POST, DELETE, OPTIONS`). ## Testing -Test the chat endpoint: ```bash +curl http://localhost:8000/health + curl -X POST http://localhost:8000/chat \ -H "Content-Type: application/json" \ - -d '{"message": "Hello!", "model": "llama3.2:latest"}' + -d '{"message": "Hello!"}' ``` -## Frontend Integration +For an end-to-end check of the whole stack use `python run_system.py --health` or +`python system_health_check.py`. There is no automated test suite in this directory. + +## Frontend integration + +The frontend reads `NEXT_PUBLIC_API_URL` (default `http://localhost:8000`) for this +server and `NEXT_PUBLIC_RAG_API_URL` (default `http://localhost:8001`) for the streaming +endpoint. Both are inlined at `next build` time, so they must be set before the frontend +is built. -Your React frontend should connect to: -- **Backend**: `http://localhost:8000` -- **Chat endpoint**: `http://localhost:8000/chat` +## Implemented today -## What's Next +- Session-scoped chat with SQLite-persisted history and auto-generated titles. +- Document upload (per session and per index) into `shared_uploads/`. +- Indexing delegated to the RAG API, with the resulting configuration stored as index + metadata. +- Vector-store bookkeeping: index records point at their LanceDB table, and deleting an + index drops that table (the late-chunk `_lc` sibling is left behind). +- Retrieval-augmented answers with source documents, plus a direct-LLM fast path for + general questions. -This simple backend is ready for: -- โœ… **Real-time chat** with local LLMs -- ๐Ÿ”œ **Document upload** for RAG -- ๐Ÿ”œ **Vector database** integration -- ๐Ÿ”œ **Streaming responses** -- ๐Ÿ”œ **Chat history** persistence \ No newline at end of file +Streaming is **not** served by this process โ€” the frontend streams from the RAG API's +`/chat/stream` endpoint directly. diff --git a/backend/chat_data.db b/backend/chat_data.db deleted file mode 100644 index 501df6be..00000000 Binary files a/backend/chat_data.db and /dev/null differ diff --git a/backend/database.py b/backend/database.py index a5d38aec..4e1f145e 100644 --- a/backend/database.py +++ b/backend/database.py @@ -1,30 +1,63 @@ +import os import sqlite3 import uuid import json from datetime import datetime from typing import List, Dict, Optional, Tuple + +def resolve_lancedb_path() -> str: + """LanceDB location, kept in sync with the store the indexing pipeline writes to.""" + env_path = os.getenv('LANCEDB_PATH') + if env_path: + return env_path + try: + from rag_system.main import PIPELINE_CONFIGS + uri = PIPELINE_CONFIGS.get('default', {}).get('storage', {}).get('lancedb_uri') + if uri: + return uri + except Exception: + pass + return './lancedb' + + class ChatDatabase: def __init__(self, db_path: str = None): + if db_path is None: + db_path = os.getenv("DB_PATH") if db_path is None: # Auto-detect environment and set appropriate path - import os if os.path.exists("/app"): # Docker environment - self.db_path = "/app/backend/chat_data.db" + db_path = "/app/backend/chat_data.db" else: # Local development environment - self.db_path = "backend/chat_data.db" - else: - self.db_path = db_path + db_path = "backend/chat_data.db" + self.db_path = db_path + parent_dir = os.path.dirname(os.path.abspath(self.db_path)) + if parent_dir: + os.makedirs(parent_dir, exist_ok=True) self.init_database() - + + def _connect(self) -> sqlite3.Connection: + """Open a connection with foreign keys enforced and a 30s busy timeout. + + Every method must go through here: PRAGMA foreign_keys is per-connection, + so a bare sqlite3.connect() silently disables the ON DELETE CASCADE rules. + The timeout rides out write locks from the other process (the gateway and + the RAG API share this SQLite file). + """ + conn = sqlite3.connect(self.db_path, timeout=30) + conn.execute("PRAGMA foreign_keys = ON") + return conn + def init_database(self): """Initialize the SQLite database with required tables""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() cursor = conn.cursor() - - # Enable foreign keys - conn.execute("PRAGMA foreign_keys = ON") - + + # WAL lets one process read while the other holds a write lock; the + # setting persists in the database file, so setting it on init is enough. + conn.execute("PRAGMA journal_mode=WAL") + # Sessions table conn.execute(''' CREATE TABLE IF NOT EXISTS sessions ( @@ -110,7 +143,7 @@ def create_session(self, title: str, model: str) -> str: session_id = str(uuid.uuid4()) now = datetime.now().isoformat() - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.execute(''' INSERT INTO sessions (id, title, created_at, updated_at, model_used) VALUES (?, ?, ?, ?, ?) @@ -123,7 +156,7 @@ def create_session(self, title: str, model: str) -> str: def get_sessions(self, limit: int = 50) -> List[Dict]: """Get all chat sessions, ordered by most recent""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.row_factory = sqlite3.Row cursor = conn.execute(''' @@ -140,7 +173,7 @@ def get_sessions(self, limit: int = 50) -> List[Dict]: def get_session(self, session_id: str) -> Optional[Dict]: """Get a specific session""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.row_factory = sqlite3.Row cursor = conn.execute(''' @@ -160,7 +193,7 @@ def add_message(self, session_id: str, content: str, sender: str, metadata: Dict now = datetime.now().isoformat() metadata_json = json.dumps(metadata or {}) - conn = sqlite3.connect(self.db_path) + conn = self._connect() # Add the message conn.execute(''' @@ -183,7 +216,7 @@ def add_message(self, session_id: str, content: str, sender: str, metadata: Dict def get_messages(self, session_id: str, limit: int = 100) -> List[Dict]: """Get all messages for a session""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.row_factory = sqlite3.Row cursor = conn.execute(''' @@ -218,7 +251,7 @@ def get_conversation_history(self, session_id: str) -> List[Dict]: def update_session_title(self, session_id: str, title: str): """Update session title""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.execute(''' UPDATE sessions SET title = ?, updated_at = ? @@ -229,7 +262,11 @@ def update_session_title(self, session_id: str, title: str): def delete_session(self, session_id: str) -> bool: """Delete a session and all its messages""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() + # messages/session_documents cascade via ON DELETE CASCADE, but the + # session_indexes foreign keys declare no cascade, so remove those + # link rows explicitly or the session delete would violate them. + conn.execute('DELETE FROM session_indexes WHERE session_id = ?', (session_id,)) cursor = conn.execute('DELETE FROM sessions WHERE id = ?', (session_id,)) deleted = cursor.rowcount > 0 conn.commit() @@ -242,7 +279,7 @@ def delete_session(self, session_id: str) -> bool: def cleanup_empty_sessions(self) -> int: """Remove sessions with no messages""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() # Find sessions with no messages cursor = conn.execute(''' @@ -253,9 +290,12 @@ def cleanup_empty_sessions(self) -> int: empty_sessions = [row[0] for row in cursor.fetchall()] - # Delete empty sessions + # Delete empty sessions (session_indexes links first: no ON DELETE + # CASCADE is declared on that table, so the link rows would otherwise + # block the session delete under enforced foreign keys) deleted_count = 0 for session_id in empty_sessions: + conn.execute('DELETE FROM session_indexes WHERE session_id = ?', (session_id,)) cursor = conn.execute('DELETE FROM sessions WHERE id = ?', (session_id,)) if cursor.rowcount > 0: deleted_count += 1 @@ -271,7 +311,7 @@ def cleanup_empty_sessions(self) -> int: def get_stats(self) -> Dict: """Get database statistics""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() # Get session count cursor = conn.execute('SELECT COUNT(*) FROM sessions') @@ -301,7 +341,7 @@ def get_stats(self) -> Dict: def add_document_to_session(self, session_id: str, file_path: str) -> int: """Adds a document file path to a session.""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() cursor = conn.execute( "INSERT INTO session_documents (session_id, file_path) VALUES (?, ?)", (session_id, file_path) @@ -314,7 +354,7 @@ def add_document_to_session(self, session_id: str, file_path: str) -> int: def get_documents_for_session(self, session_id: str) -> List[str]: """Retrieves all document file paths for a given session.""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() cursor = conn.execute( "SELECT file_path FROM session_documents WHERE session_id = ?", (session_id,) @@ -329,7 +369,7 @@ def create_index(self, name: str, description: str|None = None, metadata: dict | idx_id = str(uuid.uuid4()) created = datetime.now().isoformat() vector_table = f"text_pages_{idx_id}" - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.execute(''' INSERT INTO indexes (id, name, description, created_at, updated_at, vector_table_name, metadata) VALUES (?,?,?,?,?,?,?) @@ -340,7 +380,7 @@ def create_index(self, name: str, description: str|None = None, metadata: dict | return idx_id def get_index(self, index_id: str) -> dict | None: - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.row_factory = sqlite3.Row cur = conn.execute('SELECT * FROM indexes WHERE id=?', (index_id,)) row = cur.fetchone() @@ -356,7 +396,7 @@ def get_index(self, index_id: str) -> dict | None: return idx def list_indexes(self) -> list[dict]: - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.row_factory = sqlite3.Row rows = conn.execute('SELECT * FROM indexes').fetchall() res = [] @@ -372,27 +412,33 @@ def list_indexes(self) -> list[dict]: return res def add_document_to_index(self, index_id: str, filename: str, stored_path: str): - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.execute('INSERT INTO index_documents (index_id, original_filename, stored_path) VALUES (?,?,?)', (index_id, filename, stored_path)) conn.commit() conn.close() def link_index_to_session(self, session_id: str, index_id: str): - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.execute('INSERT INTO session_indexes (session_id, index_id, linked_at) VALUES (?,?,?)', (session_id, index_id, datetime.now().isoformat())) conn.commit() conn.close() def get_indexes_for_session(self, session_id: str) -> list[str]: - conn = sqlite3.connect(self.db_path) - cursor = conn.execute('SELECT index_id FROM session_indexes WHERE session_id=? ORDER BY linked_at', (session_id,)) + """Index IDs linked to a session, oldest link first. + + The `id` tiebreak keeps the order deterministic when two links share a + linked_at timestamp. Callers rely on this: the gateway and the RAG API + both resolve the LAST element as the session's active index. + """ + conn = self._connect() + cursor = conn.execute('SELECT index_id FROM session_indexes WHERE session_id=? ORDER BY linked_at, id', (session_id,)) ids = [r[0] for r in cursor.fetchall()] conn.close() return ids def delete_index(self, index_id: str) -> bool: """Delete an index and its related records (documents, session links). Returns True if deleted.""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() try: # Get vector table name before deletion (optional, for LanceDB cleanup) cur = conn.execute('SELECT vector_table_name FROM indexes WHERE id = ?', (index_id,)) @@ -410,24 +456,53 @@ def delete_index(self, index_id: str) -> bool: if deleted: print(f"๐Ÿ—‘๏ธ Deleted index {index_id[:8]}... and related records") - # Optional: attempt to drop LanceDB table if available if vector_table_name: + self._cleanup_index_artifacts(index_id, vector_table_name) + return deleted + + def _cleanup_index_artifacts(self, index_id: str, vector_table_name: str): + """Best-effort removal of the on-disk artifacts an index leaves behind. + + Drops the base LanceDB table plus its '
_lc' late-chunk sibling, + and removes the sidecar files written next to the index: the overview + JSONL + embedded vectors under index_store/overviews/, and the per-table + embedder marker under /table_meta/. Never raises โ€” a cleanup + failure must not turn a successful SQL delete into an error. + """ + lancedb_path = resolve_lancedb_path() + try: + from rag_system.indexing.embedders import LanceDBManager + ldb = LanceDBManager(lancedb_path) + ldb_db = ldb.db + table_names = ldb_db.table_names() if hasattr(ldb_db, 'table_names') else [] + for table in (vector_table_name, f"{vector_table_name}_lc"): try: - from rag_system.indexing.embedders import LanceDBManager - import os - db_path = os.getenv('LANCEDB_PATH') or './rag_system/index_store/lancedb' - ldb = LanceDBManager(db_path) - db = ldb.db - if hasattr(db, 'table_names') and vector_table_name in db.table_names(): - db.drop_table(vector_table_name) - print(f"๐Ÿšฎ Dropped LanceDB table '{vector_table_name}'") + if table in table_names: + ldb_db.drop_table(table) + print(f"๐Ÿšฎ Dropped LanceDB table '{table}'") except Exception as e: - print(f"โš ๏ธ Could not drop LanceDB table '{vector_table_name}': {e}") - return deleted + print(f"โš ๏ธ Could not drop LanceDB table '{table}': {e}") + except Exception as e: + print(f"โš ๏ธ Could not drop LanceDB tables for '{vector_table_name}': {e}") + + sidecars = [ + os.path.join('index_store', 'overviews', f"{index_id}.jsonl"), + os.path.join('index_store', 'overviews', f"{index_id}.vectors.npz"), + os.path.join(lancedb_path, 'table_meta', f"{vector_table_name}.json"), + os.path.join(lancedb_path, 'table_meta', f"{vector_table_name}_lc.json"), + ] + for path in sidecars: + try: + os.remove(path) + print(f"๐Ÿšฎ Removed sidecar '{path}'") + except FileNotFoundError: + pass + except OSError as e: + print(f"โš ๏ธ Could not remove sidecar '{path}': {e}") def update_index_metadata(self, index_id: str, updates: dict): """Merge new key/values into an index's metadata JSON column.""" - conn = sqlite3.connect(self.db_path) + conn = self._connect() conn.row_factory = sqlite3.Row cur = conn.execute('SELECT metadata FROM indexes WHERE id=?', (index_id,)) row = cur.fetchone() @@ -464,12 +539,10 @@ def inspect_and_populate_index_metadata(self, index_id: str) -> dict: # Try to import the RAG system modules try: from rag_system.indexing.embedders import LanceDBManager - import os - - # Use the same path as the system - db_path = os.getenv('LANCEDB_PATH') or './rag_system/index_store/lancedb' - ldb = LanceDBManager(db_path) - + + # Use the same store the indexing pipeline writes to + ldb = LanceDBManager(resolve_lancedb_path()) + # Check if table exists if not hasattr(ldb.db, 'table_names') or vector_table_name not in ldb.db.table_names(): # Table doesn't exist - this means the index was never properly built @@ -525,8 +598,14 @@ def inspect_and_populate_index_metadata(self, index_id: str) -> dict: 384: 'BAAI/bge-small-en-v1.5 (or similar)', 512: 'sentence-transformers/all-MiniLM-L6-v2 (or similar)', 768: 'BAAI/bge-base-en-v1.5 (or similar)', - 1024: 'Qwen/Qwen3-Embedding-0.6B (or similar)', - 1536: 'text-embedding-ada-002 (or similar)' + # 1024 is ambiguous by construction: harrier-oss-v1-0.6b + # (the default) and Qwen3-Embedding-0.6B share it. The + # authoritative answer is the table's own embedder marker + # (rag_system/indexing/embedders.py), not this guess. + 1024: 'microsoft/harrier-oss-v1-0.6b or Qwen/Qwen3-Embedding-0.6B (or similar)', + 1536: 'text-embedding-ada-002 (or similar)', + 2560: 'Qwen/Qwen3-Embedding-4B (or similar)', + 4096: 'Qwen/Qwen3-Embedding-8B (or similar)' } if len(vector_data) in dim_to_model: inferred_metadata['embedding_model_inferred'] = dim_to_model[len(vector_data)] @@ -671,7 +750,7 @@ def generate_session_title(first_message: str, max_length: int = 50) -> str: print("๐Ÿงช Testing database...") # Create a test session - session_id = db.create_session("Test Chat", "llama3.2:latest") + session_id = db.create_session("Test Chat", os.getenv("GENERATION_MODEL", "qwen3.5:9b")) # Add some messages db.add_message(session_id, "Hello!", "user") diff --git a/backend/ollama_client.py b/backend/ollama_client.py index 33ee103d..8dbd2d8a 100644 --- a/backend/ollama_client.py +++ b/backend/ollama_client.py @@ -1,8 +1,35 @@ import requests import json import os +import re from typing import List, Dict, Optional +DEFAULT_GENERATION_MODEL = os.getenv("GENERATION_MODEL", "qwen3.5:9b") + +# Context-window sizing: identical scheme to rag_system/utils/ollama_client.py +# (duplicated because the gateway is a standalone stdlib-only service). +_NUM_CTX_BUCKETS = (8192, 16384, 32768) +_OUTPUT_HEADROOM_TOKENS = 2048 + + +def _num_ctx_for(char_count: int) -> int: + pinned = os.getenv("OLLAMA_NUM_CTX") + if pinned: + try: + return max(2048, int(pinned)) + except ValueError: + pass + try: + max_ctx = int(os.getenv("OLLAMA_NUM_CTX_MAX", "32768")) + except ValueError: + max_ctx = 32768 + estimated = char_count // 3 + _OUTPUT_HEADROOM_TOKENS + for bucket in _NUM_CTX_BUCKETS: + if estimated <= bucket <= max_ctx: + return bucket + return max_ctx + + class OllamaClient: def __init__(self, base_url: Optional[str] = None): if base_url is None: @@ -54,21 +81,28 @@ def pull_model(self, model_name: str) -> bool: print(f"Error pulling model: {e}") return False - def chat(self, message: str, model: str = "llama3.2", conversation_history: List[Dict] = None, enable_thinking: bool = True) -> str: - """Send a chat message to Ollama""" + def chat(self, message: str, model: str = None, conversation_history: List[Dict] = None, enable_thinking: bool = True) -> str: + """Send a chat message to Ollama. + + Raises requests.exceptions.RequestException (Timeout, ConnectionError, + HTTPError) on failure โ€” the gateway maps those to 504/502 instead of + embedding error text in a 200 OK answer. + """ + if model is None: + model = DEFAULT_GENERATION_MODEL if conversation_history is None: conversation_history = [] - + # Add user message to conversation messages = conversation_history + [{"role": "user", "content": message}] - + try: payload = { "model": model, "messages": messages, "stream": False, } - + # Multiple approaches to disable thinking tokens if not enable_thinking: payload.update({ @@ -82,6 +116,15 @@ def chat(self, message: str, model: str = "llama3.2", conversation_history: List }) else: payload["think"] = True + + # Size the context window to the conversation: Ollama front-truncates + # (silently drops the OLDEST messages/system prompt) when the request + # exceeds its server-side slot, so request a window that fits. Same + # scheme as rag_system/utils/ollama_client.py: bucketed, env-capped. + payload.setdefault("options", {}).setdefault( + "num_ctx", + _num_ctx_for(sum(len(m.get("content") or "") for m in messages)), + ) response = requests.post( f"{self.api_url}/chat", @@ -92,79 +135,75 @@ def chat(self, message: str, model: str = "llama3.2", conversation_history: List if response.status_code == 200: result = response.json() response_text = result["message"]["content"] - + # Additional cleanup: remove any thinking tokens that might slip through if not enable_thinking: # Remove common thinking token patterns - import re response_text = re.sub(r'.*?', '', response_text, flags=re.DOTALL | re.IGNORECASE) response_text = re.sub(r'.*?', '', response_text, flags=re.DOTALL | re.IGNORECASE) response_text = response_text.strip() - + return response_text else: - return f"Error: {response.status_code} - {response.text}" - - except requests.exceptions.RequestException as e: - return f"Connection error: {e}" - - def chat_stream(self, message: str, model: str = "llama3.2", conversation_history: List[Dict] = None, enable_thinking: bool = True): - """Stream chat response from Ollama""" + # Not answer text: raise so the gateway maps it to a 502. + response.raise_for_status() + + except requests.exceptions.RequestException: + # Timeout / ConnectionError / HTTPError propagate to the caller. + raise + + def chat_stream(self, message: str, model: str = None, conversation_history: List[Dict] = None, enable_thinking: bool = False): + """Streaming sibling of chat(): yields response-content deltas. + + Same request shape as chat() but with stream=True; Ollama answers with + one JSON object per line, each carrying a message.content delta. + Connection-level failures (Timeout/ConnectionError/HTTPError) raise + before the first yield so the gateway can still answer 502/504; + anything after the first yield raises mid-iteration and the caller โ€” + already committed to a 200 SSE response โ€” must emit an error event. + """ + if model is None: + model = DEFAULT_GENERATION_MODEL if conversation_history is None: conversation_history = [] - + messages = conversation_history + [{"role": "user", "content": message}] - - try: - payload = { - "model": model, - "messages": messages, - "stream": True, - } - - # Multiple approaches to disable thinking tokens - if not enable_thinking: - payload.update({ - "think": False, # Native Ollama parameter - "options": { - "think": False, - "thinking": False, - "temperature": 0.7, - "top_p": 0.9 - } - }) - else: - payload["think"] = True - - response = requests.post( - f"{self.api_url}/chat", - json=payload, - stream=True, - timeout=60 - ) - - if response.status_code == 200: - for line in response.iter_lines(): - if line: - try: - data = json.loads(line) - if "message" in data and "content" in data["message"]: - content = data["message"]["content"] - - # Filter out thinking tokens in streaming mode - if not enable_thinking: - # Skip content that looks like thinking tokens - if '' in content.lower() or '' in content.lower(): - continue - - yield content - except json.JSONDecodeError: - continue - else: - yield f"Error: {response.status_code} - {response.text}" - - except requests.exceptions.RequestException as e: - yield f"Connection error: {e}" + payload = { + "model": model, + "messages": messages, + "stream": True, + } + if not enable_thinking: + payload.update({ + "think": False, + "options": {"think": False, "thinking": False, "temperature": 0.7, "top_p": 0.9}, + }) + else: + payload["think"] = True + payload.setdefault("options", {}).setdefault( + "num_ctx", + _num_ctx_for(sum(len(m.get("content") or "") for m in messages)), + ) + + response = requests.post( + f"{self.api_url}/chat", + json=payload, + timeout=60, + stream=True, + ) + response.raise_for_status() + for line in response.iter_lines(): + if not line: + continue + try: + chunk = json.loads(line) + except json.JSONDecodeError: + continue + delta = (chunk.get("message") or {}).get("content") or "" + if delta: + yield delta + if chunk.get("done"): + break def main(): """Test the Ollama client""" @@ -183,9 +222,9 @@ def main(): models = client.list_models() print(f"Available models: {models}") - # Try to use llama3.2, pull if needed - model_name = "llama3.2" - if model_name not in [m.split(":")[0] for m in models]: + # Try to use the configured generation model, pull if needed + model_name = DEFAULT_GENERATION_MODEL + if model_name not in models: print(f"Model {model_name} not found. Pulling...") if client.pull_model(model_name): print(f"โœ… Model {model_name} pulled successfully!") diff --git a/backend/requirements.txt b/backend/requirements.txt index bbd44d99..df7458c2 100644 --- a/backend/requirements.txt +++ b/backend/requirements.txt @@ -1,3 +1,2 @@ requests python-dotenv -PyPDF2 \ No newline at end of file diff --git a/backend/server.py b/backend/server.py index 859040ef..7c817805 100644 --- a/backend/server.py +++ b/backend/server.py @@ -1,10 +1,10 @@ import json import http.server import socketserver -import cgi +import email import os import uuid -from urllib.parse import urlparse, parse_qs +from urllib.parse import urlparse import requests # ๐Ÿ†• Import requests for making HTTP calls import sys from datetime import datetime @@ -14,24 +14,243 @@ # Import RAG system modules for complete metadata try: - from rag_system.main import PIPELINE_CONFIGS + from rag_system.main import PIPELINE_CONFIGS, OLLAMA_CONFIG RAG_SYSTEM_AVAILABLE = True print("โœ… RAG system modules accessible from backend") except ImportError as e: PIPELINE_CONFIGS = {} + OLLAMA_CONFIG = {} RAG_SYSTEM_AVAILABLE = False print(f"โš ๏ธ RAG system modules not available: {e}") from ollama_client import OllamaClient from database import db, generate_session_title -import simple_pdf_processor as pdf_module -from simple_pdf_processor import initialize_simple_pdf_processor -from typing import List, Dict, Any +from typing import List, Dict, Any, Optional, Tuple import re -# ๐Ÿ†• Reusable TCPServer with address reuse enabled -class ReusableTCPServer(socketserver.TCPServer): +PORT = 8000 + +# Base URL of the RAG API service. In Docker this is http://rag-api:8001. +RAG_API_URL = os.getenv("RAG_API_URL", "http://localhost:8001").rstrip("/") +RAG_API_TIMEOUT = float(os.getenv("RAG_API_TIMEOUT", "600")) +RAG_API_INDEX_TIMEOUT = float(os.getenv("RAG_API_INDEX_TIMEOUT", "3600")) + +GENERATION_MODEL = os.getenv("GENERATION_MODEL") or OLLAMA_CONFIG.get("generation_model") or "qwen3.5:9b" +ENRICHMENT_MODEL = os.getenv("ENRICHMENT_MODEL") or OLLAMA_CONFIG.get("enrichment_model") or "qwen3.5:4b" + +# Multipart uploads are parsed fully in memory, so cap the declared body size +# (default 200 MB) and answer 413 beyond it instead of reading unbounded data. +MAX_UPLOAD_BYTES = int(os.getenv("MAX_UPLOAD_BYTES", str(200 * 1024 * 1024))) + +# Canonical snake_case option name -> (caster, accepted aliases). +# The frontend historically sent camelCase, so both spellings map to one key. +CHAT_OPTIONS: Dict[str, Tuple[Any, Tuple[str, ...]]] = { + "model": (str, ()), + "compose_sub_answers": (bool, ("composeSubAnswers",)), + "query_decompose": (bool, ("queryDecompose", "decompose")), + "ai_rerank": (bool, ("aiRerank",)), + "context_expand": (bool, ("contextExpand",)), + "verify": (bool, ()), + "retrieval_k": (int, ("retrievalK",)), + "context_window_size": (int, ("contextWindowSize",)), + "reranker_top_k": (int, ("rerankerTopK",)), + "retrieval_mode": (str, ("retrievalMode", "search_type", "searchType")), + "provence_prune": (bool, ("provencePrune",)), + "provence_threshold": (float, ("provenceThreshold",)), + # Metadata filter object (roadmap 4.4). Validated by the RAG API, not here: + # dict() copies a dict and raises on anything else, so a malformed value + # still reaches the one validator and comes back as its 400. + "filters": (dict, ()), +} + +INDEX_OPTIONS: Dict[str, Tuple[Any, Tuple[str, ...]]] = { + "chunk_size": (int, ("chunkSize",)), + "window_size": (int, ("windowSize",)), + "retrieval_mode": (str, ("retrievalMode",)), + "enable_enrich": (bool, ("enableEnrich",)), + "enable_latechunk": (bool, ("enableLatechunk", "latechunk")), + "enable_docling_chunk": (bool, ("enableDoclingChunk", "doclingChunk")), + "embedding_model": (str, ("embeddingModel",)), + "enrich_model": (str, ("enrichModel",)), + "overview_model_name": (str, ("overviewModelName", "overviewModel", "overview_model")), + "batch_size_embed": (int, ("batchSizeEmbed",)), + "batch_size_enrich": (int, ("batchSizeEnrich",)), +} + + +def normalize_options(data: dict, spec: Dict[str, Tuple[Any, Tuple[str, ...]]]) -> Dict[str, Any]: + """Return canonical snake_case options from a body that may use either casing.""" + options: Dict[str, Any] = {} + for canonical, (caster, aliases) in spec.items(): + for key in (canonical,) + tuple(aliases): + if key not in data or data[key] is None: + continue + try: + options[canonical] = caster(data[key]) + except (TypeError, ValueError): + options[canonical] = data[key] + break + return options + + +class ChatBackendError(Exception): + """An upstream chat failure (RAG API or Ollama) carrying its HTTP status. + + Raised instead of embedding failure text in a 200 OK answer, so clients + can distinguish an error from a real response: timeout โ†’ 504, unreachable + or non-200 upstream โ†’ 502. + """ + def __init__(self, message: str, status_code: int): + super().__init__(message) + self.status_code = status_code + + +# --------------------------------------------------------------------------- +# Gateway routing gate +# --------------------------------------------------------------------------- +# Retrieval-first cascade: escalate, don't pre-decide. When a session has +# documents linked, the gateway sends the message to the RAG API unless it is +# unmistakable smalltalk or a question about the assistant itself. There is no +# LLM call here โ€” pre-retrieval LLM routing is measurably the weakest routing +# pattern available (see Documentation/research/), and the agent-side triage in +# rag_system/agent/loop.py remains the single LLM routing layer: it can still +# answer directly, so over-sending to RAG here is cheap and recoverable. +# +# Both regexes below are whole-message allowlists. Anything that is not an +# exact match falls through to RAG. + +SMALLTALK_MAX_WORDS = 6 + +# Phrases that, on their own, carry no retrievable intent. +_SMALLTALK_CORE = ( + # greetings + r"hi+", r"hey+", r"hello+", r"heya", r"yo", r"howdy", r"greetings", + r"good\s+(?:morning|afternoon|evening|day)", + r"how\s+are\s+(?:you|u|ya)(?:\s+doing)?", r"how'?s\s+it\s+going", + r"what'?s\s+up", r"sup", + # thanks + r"thanks", r"thank\s+you", r"thx", r"ty", r"cheers", + r"much\s+appreciated", r"appreciate\s+it", + # farewells + r"bye+", r"goodbye", r"good\s*night", r"see\s+(?:you|ya)", + r"talk\s+to\s+you\s+later", r"take\s+care", r"later", + # acknowledgements / fillers that end a turn + r"ok(?:ay)?", r"kk", r"alright", r"got\s+it", r"understood", r"i\s+see", + r"sounds\s+good", r"no\s+problem", r"np", r"never\s*mind", r"nvm", + r"cool", r"nice", r"awesome", r"perfect", r"great", + r"yes", r"yeah", r"yep", r"nope", r"no", + r"sorry", r"please", r"lol", r"haha+", +) + +# Words allowed *alongside* a core phrase but never sufficient on their own. +_SMALLTALK_FILLER = ( + r"there", r"again", r"all", r"everyone", r"folks", r"guys", r"team", + r"friend", r"buddy", r"mate", r"man", r"dude", r"bot", r"assistant", + r"a\s+lot", r"so\s+much", r"very\s+much", r"so", r"much", r"very", + r"then", r"too", r"and", r"well", r"my", r"you", +) + + +def _alternation(*groups: Tuple[str, ...]) -> str: + """Join regex phrases longest-first so the widest match is tried first.""" + phrases = [p for group in groups for p in group] + phrases.sort(key=len, reverse=True) + return "|".join(phrases) + + +# Whole message consists only of allowlisted smalltalk/filler phrases. +_SMALLTALK_RE = re.compile( + r"^\W*(?:{alt})(?:\W+(?:{alt}))*\W*$".format( + alt=_alternation(_SMALLTALK_CORE, _SMALLTALK_FILLER) + ), + re.IGNORECASE, +) + +# ...and at least one of those phrases is a *core* smalltalk phrase. +_SMALLTALK_CORE_RE = re.compile( + r"(? bool: + """True when a message is pure smalltalk or a question about the assistant. + + Deterministic and allocation-cheap: two anchored regexes and a word count. + Deliberately conservative โ€” an unmatched message routes to RAG. + """ + text = (message or "").strip() + if not text: + return True + if _ASSISTANT_META_RE.match(text): + return True + if len(text.split()) > SMALLTALK_MAX_WORDS: + return False + return bool(_SMALLTALK_RE.match(text) and _SMALLTALK_CORE_RE.search(text)) + + +def should_use_rag(message: str, idx_ids: Optional[List[str]], force_rag: bool = False) -> bool: + """Decide whether one chat message goes to the RAG API or straight to Ollama. + + 1. ``force_rag`` โ†’ RAG, unconditionally. + 2. No indexes linked to the session โ†’ direct LLM (nothing to retrieve from). + 3. Smalltalk / assistant-meta โ†’ direct LLM. + 4. Everything else โ†’ RAG. + """ + if force_rag: + return True + if not idx_ids: + return False + return not is_smalltalk_or_meta(message) + + +def default_index_metadata() -> Dict[str, Any]: + """Index metadata defaults taken from the live RAG pipeline configuration.""" + config = PIPELINE_CONFIGS.get('default', {}) if RAG_SYSTEM_AVAILABLE else {} + retrieval = config.get('retrieval', {}) + indexing = config.get('indexing', {}) + return { + 'chunk_size': 512, + 'retrieval_mode': retrieval.get('search_type', 'hybrid'), + 'window_size': config.get('contextual_enricher', {}).get('window_size', 1), + 'embedding_model': os.getenv('EMBEDDING_MODEL') or config.get('embedding_model_name') or 'microsoft/harrier-oss-v1-0.6b', + 'enrich_model': ENRICHMENT_MODEL, + 'overview_model': ENRICHMENT_MODEL, + 'enable_enrich': config.get('contextual_enricher', {}).get('enabled', True), + 'latechunk': retrieval.get('latechunk', {}).get('enabled', False), + 'docling_chunk': True, + 'batch_size_embed': indexing.get('embedding_batch_size', 50), + 'batch_size_enrich': indexing.get('enrichment_batch_size', 25), + } + + +# ๐Ÿ†• Threaded TCPServer with address reuse enabled. Threading keeps a slow RAG +# query from blocking every other request, including the RAG API's callback. +class ReusableTCPServer(socketserver.ThreadingTCPServer): allow_reuse_address = True + daemon_threads = True class ChatHandler(http.server.BaseHTTPRequestHandler): def __init__(self, *args, **kwargs): @@ -78,8 +297,7 @@ def do_GET(self): session_id = parsed_path.path.split('/')[-1] self.handle_get_session(session_id) else: - self.send_response(404) - self.end_headers() + self.send_json_response({"error": "Not found"}, status_code=404) def do_POST(self): """Handle POST requests""" @@ -87,6 +305,8 @@ def do_POST(self): if parsed_path.path == '/chat': self.handle_chat() + elif parsed_path.path == '/chat/stream': + self.handle_chat_stream() elif parsed_path.path == '/sessions': self.handle_create_session() elif parsed_path.path == '/indexes': @@ -99,9 +319,17 @@ def do_POST(self): self.handle_build_index(index_id) elif parsed_path.path.startswith('/sessions/') and '/indexes/' in parsed_path.path: parts = parsed_path.path.split('/') + if len(parts) != 5 or parts[3] != 'indexes' or not parts[2] or not parts[4]: + self.send_json_response({ + "error": "Malformed path: expected /sessions//indexes/" + }, status_code=400) + return session_id = parts[2] index_id = parts[4] self.handle_link_index_to_session(session_id, index_id) + elif parsed_path.path.startswith('/sessions/') and parsed_path.path.endswith('/messages/save'): + session_id = parsed_path.path.split('/')[-3] + self.handle_save_messages(session_id) elif parsed_path.path.startswith('/sessions/') and parsed_path.path.endswith('/messages'): session_id = parsed_path.path.split('/')[-2] self.handle_session_chat(session_id) @@ -115,8 +343,7 @@ def do_POST(self): session_id = parsed_path.path.split('/')[-2] self.handle_rename_session(session_id) else: - self.send_response(404) - self.end_headers() + self.send_json_response({"error": "Not found"}, status_code=404) def do_DELETE(self): """Handle DELETE requests""" @@ -129,18 +356,17 @@ def do_DELETE(self): index_id = parsed_path.path.split('/')[-1] self.handle_delete_index(index_id) else: - self.send_response(404) - self.end_headers() + self.send_json_response({"error": "Not found"}, status_code=404) def handle_chat(self): """Handle legacy chat requests (without sessions)""" try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - + data = self.read_required_json_body() + if data is None: + return + message = data.get('message', '') - model = data.get('model', 'llama3.2:latest') + model = data.get('model', GENERATION_MODEL) conversation_history = data.get('conversation_history', []) if not message: @@ -169,11 +395,105 @@ def handle_chat(self): self.send_json_response({ "error": "Invalid JSON" }, status_code=400) + except requests.exceptions.Timeout: + self.send_json_response({ + "error": "The Ollama backend did not respond in time." + }, status_code=504) + except requests.exceptions.ConnectionError: + self.send_json_response({ + "error": "Could not connect to Ollama. Please ensure it is running." + }, status_code=502) + except requests.exceptions.HTTPError as e: + status = e.response.status_code if e.response is not None else "unknown" + self.send_json_response({ + "error": f"Ollama request failed (HTTP {status})." + }, status_code=502) except Exception as e: self.send_json_response({ "error": f"Server error: {str(e)}" }, status_code=500) + def handle_chat_stream(self): + """SSE variant of /chat: stream token deltas from Ollama. + + Same event shape as the RAG API's /chat/stream ("token" / "complete" / + "error" objects on data: lines) so the frontend shares one parser. + Pre-stream failures still answer real HTTP errors; once the SSE header + is out, failures become a terminal "error" event. + """ + try: + data = self.read_required_json_body() + if data is None: + return + + message = data.get('message', '') + model = data.get('model', GENERATION_MODEL) + conversation_history = data.get('conversation_history', []) + + if not message: + self.send_json_response({"error": "Message is required"}, status_code=400) + return + if not self.ollama_client.is_ollama_running(): + self.send_json_response({ + "error": "Ollama is not running. Please start Ollama first." + }, status_code=503) + return + + try: + token_iter = self.ollama_client.chat_stream(message, model, conversation_history) + first = next(token_iter, None) + except requests.exceptions.Timeout: + self.send_json_response({"error": "The Ollama backend did not respond in time."}, status_code=504) + return + except requests.exceptions.ConnectionError: + self.send_json_response({"error": "Could not connect to Ollama. Please ensure it is running."}, status_code=502) + return + except requests.exceptions.HTTPError as e: + status = e.response.status_code if e.response is not None else "unknown" + self.send_json_response({"error": f"Ollama request failed (HTTP {status})."}, status_code=502) + return + + self.send_response(200) + self.send_header('Content-Type', 'text/event-stream') + self.send_header('Cache-Control', 'no-cache') + self.send_header('Connection', 'close') + self.send_header('Access-Control-Allow-Origin', '*') + self.send_header('Access-Control-Allow-Methods', 'GET, POST, DELETE, OPTIONS') + self.send_header('Access-Control-Allow-Headers', 'Content-Type') + self.end_headers() + + def emit(evt_type, payload): + frame = f"data: {json.dumps({'type': evt_type, 'data': payload})}\n\n" + self.wfile.write(frame.encode('utf-8')) + self.wfile.flush() + + parts = [] + try: + if first: + parts.append(first) + emit('token', {'text': first}) + for delta in token_iter: + parts.append(delta) + emit('token', {'text': delta}) + emit('complete', { + 'response': ''.join(parts), + 'model': model, + 'message_count': len(conversation_history) + 1, + }) + except (BrokenPipeError, ConnectionResetError): + return # client went away mid-stream; nothing to answer + except Exception as e: + # Mid-stream upstream failure: the 200 is already committed, so + # report through the stream like the RAG API does. + try: + emit('error', {'error': f'Ollama stream failed: {e}'}) + except (BrokenPipeError, ConnectionResetError): + pass + except json.JSONDecodeError: + self.send_json_response({"error": "Invalid JSON"}, status_code=400) + except Exception as e: + self.send_json_response({"error": f"Server error: {str(e)}"}, status_code=500) + def handle_get_sessions(self): """Get all chat sessions""" try: @@ -245,13 +565,13 @@ def handle_get_session_documents(self, session_id: str): def handle_create_session(self): """Create a new chat session""" try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - + data = self.read_required_json_body() + if data is None: + return + title = data.get('title', 'New Chat') - model = data.get('model', 'llama3.2:latest') - + model = data.get('model', GENERATION_MODEL) + session_id = db.create_session(title, model) session = db.get_session(session_id) @@ -280,9 +600,9 @@ def handle_session_chat(self, session_id: str): self.send_json_response({"error": "Session not found"}, status_code=404) return - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) + data = self.read_required_json_body() + if data is None: + return message = data.get('message', '') if not message: @@ -295,23 +615,37 @@ def handle_session_chat(self, session_id: str): # Add user message to database first user_message_id = db.add_message(session_id, message, "user") - - # ๐ŸŽฏ SMART ROUTING: Decide between direct LLM vs RAG + + options = normalize_options(data, CHAT_OPTIONS) + + # ๐ŸŽฏ ROUTING: deterministic gate, no LLM call (see should_use_rag) idx_ids = db.get_indexes_for_session(session_id) - force_rag = bool(data.get("force_rag", False)) - use_rag = True if force_rag else self._should_use_rag(message, idx_ids) - + force_rag = bool(data.get("force_rag", data.get("forceRag", False))) + if force_rag: + options["force_rag"] = True + # An explicit metadata filter (roadmap 4.4) is a statement that this + # is a document question โ€” gateway twin of the agent-side rule. + if options.get("filters"): + force_rag = True + use_rag = should_use_rag(message, idx_ids, force_rag=force_rag) + if use_rag: # ๐Ÿ” --- Use RAG Pipeline for Document-Related Queries --- print(f"๐Ÿ” Using RAG pipeline for document query: '{message[:50]}...'") - response_text, source_docs = self._handle_rag_query(session_id, message, data, idx_ids) + response_text, source_docs = self._handle_rag_query(session_id, message, options, idx_ids) else: # โšก --- Use Direct LLM for General Queries (FAST) --- print(f"โšก Using direct LLM for general query: '{message[:50]}...'") - response_text, source_docs = self._handle_direct_llm_query(session_id, message, session) - - # Add AI response to database - ai_message_id = db.add_message(session_id, response_text, "assistant") + response_text, source_docs = self._handle_direct_llm_query( + session_id, message, session, options.get('model') + ) + + # Add AI response to database (sources go into metadata so reloaded + # sessions can still render attribution) + ai_message_id = db.add_message( + session_id, response_text, "assistant", + metadata={'source_documents': source_docs} if source_docs else None + ) updated_session = db.get_session(session_id) @@ -325,11 +659,16 @@ def handle_session_chat(self, session_id: str): except BrokenPipeError: # Client disconnected - this is normal for long queries, just log it - print(f"โš ๏ธ Client disconnected during RAG processing for query: '{message[:30]}...'") + preview = message[:30] if 'message' in locals() else '' + print(f"โš ๏ธ Client disconnected during RAG processing for query: '{preview}...'") except json.JSONDecodeError: self.send_json_response({ "error": "Invalid JSON" }, status_code=400) + except ChatBackendError as e: + # Upstream RAG/direct-LLM failure: answer with a real error status + # so clients can distinguish failure from an answer. + self.send_json_response({"error": str(e)}, status_code=e.status_code) except Exception as e: print(f"โŒ Server error in session chat: {str(e)}") try: @@ -339,232 +678,24 @@ def handle_session_chat(self, session_id: str): except BrokenPipeError: print(f"โš ๏ธ Client disconnected during error response") - def _should_use_rag(self, message: str, idx_ids: List[str]) -> bool: - """ - ๐Ÿง  ENHANCED: Determine if a query should use RAG pipeline using document overviews. - - Args: - message: The user's query - idx_ids: List of index IDs associated with the session - - Returns: - bool: True if should use RAG, False for direct LLM - """ - # No indexes = definitely no RAG needed - if not idx_ids: - return False - - # Load document overviews for intelligent routing - try: - doc_overviews = self._load_document_overviews(idx_ids) - if doc_overviews: - return self._route_using_overviews(message, doc_overviews) - except Exception as e: - print(f"โš ๏ธ Overview-based routing failed, falling back to simple routing: {e}") - - # Fallback to simple pattern matching if overviews unavailable - return self._simple_pattern_routing(message, idx_ids) - - def _load_document_overviews(self, idx_ids: List[str]) -> List[str]: - """Load and aggregate overviews for the given index IDs. - - Strategy: - 1. Attempt to load each index's dedicated overview file. - 2. Aggregate all overviews found across available files (deduplicated). - 3. If none of the index files exist, fall back to the legacy global overview file. - """ - import os, json - - aggregated: list[str] = [] - - # 1๏ธโƒฃ Collect overviews from per-index files - for idx in idx_ids: - candidate_paths = [ - f"../index_store/overviews/{idx}.jsonl", - f"index_store/overviews/{idx}.jsonl", - f"./index_store/overviews/{idx}.jsonl", - ] - for p in candidate_paths: - if os.path.exists(p): - print(f"๐Ÿ“– Loading overviews from: {p}") - try: - with open(p, "r", encoding="utf-8") as f: - for line in f: - if not line.strip(): - continue - try: - record = json.loads(line) - overview = record.get("overview", "").strip() - if overview: - aggregated.append(overview) - except json.JSONDecodeError: - continue # skip malformed lines - break # Stop after the first existing path for this idx - except Exception as e: - print(f"โš ๏ธ Error reading {p}: {e}") - break # Don't keep trying other paths for this idx if read failed - - # 2๏ธโƒฃ Fall back to legacy global file if no per-index overviews found - if not aggregated: - legacy_paths = [ - "../index_store/overviews/overviews.jsonl", - "index_store/overviews/overviews.jsonl", - "./index_store/overviews/overviews.jsonl", - ] - for p in legacy_paths: - if os.path.exists(p): - print(f"โš ๏ธ Falling back to legacy overviews file: {p}") - try: - with open(p, "r", encoding="utf-8") as f: - for line in f: - if not line.strip(): - continue - try: - record = json.loads(line) - overview = record.get("overview", "").strip() - if overview: - aggregated.append(overview) - except json.JSONDecodeError: - continue - except Exception as e: - print(f"โš ๏ธ Error reading legacy overviews file {p}: {e}") - break - - # Limit for performance - if aggregated: - print(f"โœ… Loaded {len(aggregated)} document overviews from {len(idx_ids)} index(es)") - else: - print(f"โš ๏ธ No overviews found for indices {idx_ids}") - return aggregated[:40] - - def _route_using_overviews(self, query: str, overviews: List[str]) -> bool: - """ - ๐ŸŽฏ Use document overviews and LLM to make intelligent routing decisions. - - Returns True if RAG should be used, False for direct LLM. - """ - if not overviews: - return False - - # Format overviews for the routing prompt - overviews_block = "\n".join(f"[{i+1}] {ov}" for i, ov in enumerate(overviews)) - - router_prompt = f"""You are an AI router deciding whether a user question should be answered via: -โ€ข "USE_RAG" โ€“ search the user's private documents (described below) -โ€ข "DIRECT_LLM" โ€“ reply from general knowledge (greetings, public facts, unrelated topics) - -CRITICAL PRINCIPLE: When documents exist in the KB, strongly prefer USE_RAG unless the query is purely conversational or completely unrelated to any possible document content. - -RULES: -1. If ANY overview clearly relates to the question (entities, numbers, addresses, dates, amounts, companies, technical terms) โ†’ USE_RAG -2. For document operations (summarize, analyze, explain, extract, find) โ†’ USE_RAG -3. For greetings only ("Hi", "Hello", "Thanks") โ†’ DIRECT_LLM -4. For pure math/world knowledge clearly unrelated to documents โ†’ DIRECT_LLM -5. When in doubt โ†’ USE_RAG - -DOCUMENT OVERVIEWS: -{overviews_block} - -DECISION EXAMPLES: -โ€ข "What invoice amounts are mentioned?" โ†’ USE_RAG (document-specific) -โ€ข "Who is PromptX AI LLC?" โ†’ USE_RAG (entity in documents) -โ€ข "What is the DeepSeek model?" โ†’ USE_RAG (mentioned in documents) -โ€ข "Summarize the research paper" โ†’ USE_RAG (document operation) -โ€ข "What is 2+2?" โ†’ DIRECT_LLM (pure math) -โ€ข "Hi there" โ†’ DIRECT_LLM (greeting only) - -USER QUERY: "{query}" - -Respond with exactly one word: USE_RAG or DIRECT_LLM""" - - try: - # Use Ollama to make the routing decision - response = self.ollama_client.chat( - message=router_prompt, - model="qwen3:0.6b", # Fast model for routing - enable_thinking=False # Fast routing - ) - - # The response is directly the text, not a dict - decision = response.strip().upper() - - # Parse decision - if "USE_RAG" in decision: - print(f"๐ŸŽฏ Overview-based routing: USE_RAG for query: '{query[:50]}...'") - return True - elif "DIRECT_LLM" in decision: - print(f"โšก Overview-based routing: DIRECT_LLM for query: '{query[:50]}...'") - return False - else: - print(f"โš ๏ธ Unclear routing decision '{decision}', defaulting to RAG") - return True # Default to RAG when uncertain - - except Exception as e: - print(f"โŒ LLM routing failed: {e}, falling back to pattern matching") - return self._simple_pattern_routing(query, []) - - def _simple_pattern_routing(self, message: str, idx_ids: List[str]) -> bool: - """ - ๐Ÿ“ FALLBACK: Simple pattern-based routing (original logic). - """ - message_lower = message.lower() - - # Always use Direct LLM for greetings and casual conversation - greeting_patterns = [ - 'hello', 'hi', 'hey', 'greetings', 'good morning', 'good afternoon', 'good evening', - 'how are you', 'how do you do', 'nice to meet', 'pleasure to meet', - 'thanks', 'thank you', 'bye', 'goodbye', 'see you', 'talk to you later', - 'test', 'testing', 'check', 'ping', 'just saying', 'nevermind', - 'ok', 'okay', 'alright', 'got it', 'understood', 'i see' - ] - - # Check for greeting patterns - for pattern in greeting_patterns: - if pattern in message_lower: - return False # Use Direct LLM for greetings - - # Keywords that strongly suggest document-related queries - rag_indicators = [ - 'document', 'doc', 'file', 'pdf', 'text', 'content', 'page', - 'according to', 'based on', 'mentioned', 'states', 'says', - 'what does', 'summarize', 'summary', 'analyze', 'analysis', - 'quote', 'citation', 'reference', 'source', 'evidence', - 'explain from', 'extract', 'find in', 'search for' - ] - - # Check for strong RAG indicators - for indicator in rag_indicators: - if indicator in message_lower: - return True - - # Question words + substantial length might benefit from RAG - question_words = ['what', 'how', 'when', 'where', 'why', 'who', 'which'] - starts_with_question = any(message_lower.startswith(word) for word in question_words) - - if starts_with_question and len(message) > 40: - return True - - # Very short messages - use direct LLM - if len(message.strip()) < 20: - return False - - # Default to Direct LLM unless there's clear indication of document query - return False - - def _handle_direct_llm_query(self, session_id: str, message: str, session: dict): + def _handle_direct_llm_query(self, session_id: str, message: str, session: dict, model: Optional[str] = None): """ Handle query using direct Ollama client with thinking disabled for speed. - + Returns: tuple: (response_text, empty_source_docs) + + Raises: + ChatBackendError: on Ollama timeout (504) or connection/non-200 + failure (502), so the chat handler answers with a real error status. """ try: # Get conversation history for context conversation_history = db.get_conversation_history(session_id) - - # Use the session's model or default - model = session.get('model', 'qwen3:8b') # Default to fast model - + + # Per-request override wins, then the session's model, then the configured default + model = model or session.get('model') or GENERATION_MODEL + # Direct Ollama call with thinking disabled for speed response_text = self.ollama_client.chat( message=message, @@ -572,26 +703,42 @@ def _handle_direct_llm_query(self, session_id: str, message: str, session: dict) conversation_history=conversation_history, enable_thinking=False # โšก DISABLE THINKING FOR SPEED ) - + return response_text, [] # No source docs for direct LLM - + + except requests.exceptions.Timeout: + print("โŒ Direct LLM request to Ollama timed out.") + raise ChatBackendError("The Ollama backend did not respond in time.", 504) + except requests.exceptions.ConnectionError: + print("โŒ Could not connect to Ollama.") + raise ChatBackendError("Could not connect to Ollama. Please ensure it is running.", 502) + except requests.exceptions.HTTPError as e: + status = e.response.status_code if e.response is not None else "unknown" + print(f"โŒ Ollama returned HTTP {status}.") + raise ChatBackendError(f"Ollama request failed (HTTP {status}).", 502) except Exception as e: print(f"โŒ Direct LLM error: {e}") return f"Error processing query: {str(e)}", [] - def _handle_rag_query(self, session_id: str, message: str, data: dict, idx_ids: List[str]): + def _handle_rag_query(self, session_id: str, message: str, options: Dict[str, Any], idx_ids: List[str]): """ - Handle query using the full RAG pipeline (delegates to the advanced RAG API running on port 8001). + Handle query using the full RAG pipeline (delegates to the RAG API at RAG_API_URL). Returns: tuple[str, List[dict]]: (response_text, source_documents) + + Raises: + ChatBackendError: on RAG API timeout (504), connection failure or + non-200 response (502) โ€” never embedded in a 200 OK answer. """ # Defaults response_text = "" source_docs: List[dict] = [] - # Build payload for RAG API - rag_api_url = "http://localhost:8001/chat" + # Build payload for RAG API. The active index is the LAST linked one โ€” + # same convention as the RAG API's table/embedder resolution + # (see ChatDatabase.get_indexes_for_session). + rag_api_url = f"{RAG_API_URL}/chat" table_name = f"text_pages_{idx_ids[-1]}" if idx_ids else None payload: Dict[str, Any] = { "query": message, @@ -600,47 +747,28 @@ def _handle_rag_query(self, session_id: str, message: str, data: dict, idx_ids: if table_name: payload["table_name"] = table_name - # Copy optional parameters from the incoming request - optional_params: Dict[str, tuple[type, str]] = { - "compose_sub_answers": (bool, "compose_sub_answers"), - "query_decompose": (bool, "query_decompose"), - "ai_rerank": (bool, "ai_rerank"), - "context_expand": (bool, "context_expand"), - "verify": (bool, "verify"), - "retrieval_k": (int, "retrieval_k"), - "context_window_size": (int, "context_window_size"), - "reranker_top_k": (int, "reranker_top_k"), - "search_type": (str, "search_type"), - "dense_weight": (float, "dense_weight"), - "provence_prune": (bool, "provence_prune"), - "provence_threshold": (float, "provence_threshold"), - } - for key, (caster, payload_key) in optional_params.items(): - val = data.get(key) - if val is not None: - try: - payload[payload_key] = caster(val) # type: ignore[arg-type] - except Exception: - payload[payload_key] = val + payload.update(options) try: - rag_response = requests.post(rag_api_url, json=payload) - if rag_response.status_code == 200: - rag_data = rag_response.json() - response_text = rag_data.get("answer", "No answer found.") - source_docs = rag_data.get("source_documents", []) - else: - response_text = f"Error from RAG API ({rag_response.status_code}): {rag_response.text}" - print(f"โŒ RAG API error: {response_text}") + rag_response = requests.post(rag_api_url, json=payload, timeout=RAG_API_TIMEOUT) + except requests.exceptions.Timeout: + print(f"โŒ RAG API request timed out after {RAG_API_TIMEOUT:.0f}s ({rag_api_url}).") + raise ChatBackendError(f"The RAG API did not respond within {RAG_API_TIMEOUT:.0f}s.", 504) except requests.exceptions.ConnectionError: - response_text = "Could not connect to the RAG API server. Please ensure it is running." - print("โŒ Connection to RAG API failed (port 8001).") - except Exception as e: - response_text = f"Error processing RAG query: {str(e)}" - print(f"โŒ RAG processing error: {e}") + print(f"โŒ Connection to RAG API failed ({rag_api_url}).") + raise ChatBackendError(f"Could not connect to the RAG API server at {RAG_API_URL}. Please ensure it is running.", 502) + + if rag_response.status_code == 200: + rag_data = rag_response.json() + response_text = rag_data.get("answer", "No answer found.") + source_docs = rag_data.get("source_documents", []) + else: + # A non-200 is an upstream failure, not an answer. + print(f"โŒ RAG API error ({rag_response.status_code}): {rag_response.text}") + raise ChatBackendError(f"RAG API request failed (HTTP {rag_response.status_code}).", 502) # Strip any / tags that might slip through - response_text = re.sub(r'<(think|thinking)>.*?', '', response_text, flags=re.DOTALL | re.IGNORECASE).strip() + response_text = re.sub(r'<(think|thinking)>.*?', '', response_text, flags=re.DOTALL | re.IGNORECASE).strip() return response_text, source_docs @@ -655,36 +783,69 @@ def handle_delete_session(self, session_id: str): except Exception as e: self.send_json_response({'error': str(e)}, status_code=500) + def parse_multipart_files(self, field_name: str = 'files') -> Optional[List[Tuple[str, bytes]]]: + """Parse a multipart/form-data body and return (filename, content) for one field. + + Uses the stdlib email parser; `cgi` was removed from Python 3.13. + Returns None after sending a 413 response when the declared body + exceeds MAX_UPLOAD_BYTES; callers must treat None as "response sent". + """ + content_type = self.headers.get('Content-Type', '') or '' + if not content_type.lower().startswith('multipart/form-data'): + return [] + + length = int(self.headers.get('Content-Length', 0) or 0) + if length <= 0: + return [] + if length > MAX_UPLOAD_BYTES: + self.send_json_response({ + "error": f"Upload too large: body of {length} bytes exceeds the {MAX_UPLOAD_BYTES}-byte limit" + }, status_code=413) + return None + + body = self.rfile.read(length) + prologue = f"Content-Type: {content_type}\r\nMIME-Version: 1.0\r\n\r\n".encode('utf-8') + message = email.message_from_bytes(prologue + body) + if not message.is_multipart(): + return [] + + files: List[Tuple[str, bytes]] = [] + for part in message.walk(): + if part.is_multipart(): + continue + filename = part.get_filename() + if not filename: + continue + if field_name and part.get_param('name', header='content-disposition') != field_name: + continue + payload = part.get_payload(decode=True) + if payload is None: + continue + files.append((os.path.basename(filename), payload)) + return files + def handle_file_upload(self, session_id: str): """Handle file uploads, save them, and associate with the session.""" - form = cgi.FieldStorage( - fp=self.rfile, - headers=self.headers, - environ={'REQUEST_METHOD': 'POST', 'CONTENT_TYPE': self.headers['Content-Type']} - ) - uploaded_files = [] - if 'files' in form: - files = form['files'] - if not isinstance(files, list): - files = [files] - + incoming = self.parse_multipart_files('files') + if incoming is None: + return # 413 already sent + if incoming: upload_dir = "shared_uploads" os.makedirs(upload_dir, exist_ok=True) - for file_item in files: - if file_item.filename: - # Create a unique filename to avoid overwrites - unique_filename = f"{uuid.uuid4()}_{file_item.filename}" - file_path = os.path.join(upload_dir, unique_filename) - - with open(file_path, 'wb') as f: - f.write(file_item.file.read()) - - # Store the absolute path for the indexing service - absolute_file_path = os.path.abspath(file_path) - db.add_document_to_session(session_id, absolute_file_path) - uploaded_files.append({"filename": file_item.filename, "stored_path": absolute_file_path}) + for filename, content in incoming: + # Create a unique filename to avoid overwrites + unique_filename = f"{uuid.uuid4()}_{filename}" + file_path = os.path.join(upload_dir, unique_filename) + + with open(file_path, 'wb') as f: + f.write(content) + + # Store the absolute path for the indexing service + absolute_file_path = os.path.abspath(file_path) + db.add_document_to_session(session_id, absolute_file_path) + uploaded_files.append({"filename": filename, "stored_path": absolute_file_path}) if not uploaded_files: self.send_json_response({"error": "No files were uploaded"}, status_code=400) @@ -695,53 +856,90 @@ def handle_file_upload(self, session_id: str): "uploaded_files": uploaded_files }) + def read_json_body(self) -> dict: + """Read and decode an optional JSON request body. Returns {} when absent.""" + length = int(self.headers.get('Content-Length', 0) or 0) + if length <= 0: + return {} + body = self.rfile.read(length) + try: + parsed = json.loads(body.decode('utf-8')) + except (ValueError, UnicodeDecodeError): + return {} + return parsed if isinstance(parsed, dict) else {} + + def read_required_json_body(self) -> Optional[dict]: + """Read a required JSON request body. + + Returns None after sending a 400 when Content-Length is missing or + invalid; a JSONDecodeError still propagates so callers keep their + existing "Invalid JSON" 400 handling. + """ + try: + content_length = int(self.headers.get('Content-Length') or 0) + except (TypeError, ValueError): + self.send_json_response({"error": "Missing or invalid Content-Length"}, status_code=400) + return None + if content_length <= 0: + self.send_json_response({"error": "Request body required"}, status_code=400) + return None + return json.loads(self.rfile.read(content_length).decode('utf-8')) + def handle_index_documents(self, session_id: str): """Triggers indexing for all documents in a session.""" print(f"๐Ÿ”ฅ Received request to index documents for session {session_id[:8]}...") try: + options = normalize_options(self.read_json_body(), INDEX_OPTIONS) + file_paths = db.get_documents_for_session(session_id) if not file_paths: self.send_json_response({"message": "No documents to index for this session."}, status_code=200) return print(f"Found {len(file_paths)} documents to index. Sending to RAG API...") - - rag_api_url = "http://localhost:8001/index" - rag_response = requests.post(rag_api_url, json={"file_paths": file_paths, "session_id": session_id}) + + rag_api_url = f"{RAG_API_URL}/index" + payload: Dict[str, Any] = {"file_paths": file_paths, "session_id": session_id} + payload.update(options) + rag_response = requests.post(rag_api_url, json=payload, timeout=RAG_API_INDEX_TIMEOUT) if rag_response.status_code == 200: print("โœ… RAG API successfully indexed documents.") - # Merge key config values into index metadata - idx_meta = { + # Merge key config values into index metadata. Session-scoped + # builds name the table after the chat session, which has no + # row in the indexes table โ€” only merge when the id is a real + # index, otherwise the update would raise "Index not found". + idx_meta: Dict[str, Any] = { "session_linked": True, - "retrieval_mode": "hybrid", + "retrieval_mode": options.get("retrieval_mode", "hybrid"), } - try: - db.update_index_metadata(session_id, idx_meta) # session_id used as index_id in text table naming - except Exception as e: - print(f"โš ๏ธ Failed to update index metadata for session index: {e}") + idx_meta.update({k: v for k, v in options.items() if k != "retrieval_mode"}) + if db.get_index(session_id): + try: + db.update_index_metadata(session_id, idx_meta) + except Exception as e: + print(f"โš ๏ธ Failed to update index metadata for session index: {e}") + else: + print(f"โ„น๏ธ Skipping index metadata update: {session_id[:8]}... is a session, not an index") self.send_json_response(rag_response.json()) else: error_info = rag_response.text print(f"โŒ RAG API indexing failed ({rag_response.status_code}): {error_info}") self.send_json_response({"error": f"Indexing failed: {error_info}"}, status_code=500) + except requests.exceptions.Timeout: + print(f"โŒ RAG API indexing timed out after {RAG_API_INDEX_TIMEOUT:.0f}s.") + self.send_json_response({ + "error": f"Indexing did not complete within {RAG_API_INDEX_TIMEOUT:.0f}s." + }, status_code=504) + except requests.exceptions.ConnectionError: + print(f"โŒ Connection to RAG API failed ({RAG_API_URL}).") + self.send_json_response({ + "error": f"Could not connect to the RAG API server at {RAG_API_URL}." + }, status_code=502) except Exception as e: print(f"โŒ Exception during indexing: {str(e)}") self.send_json_response({"error": f"An unexpected error occurred: {str(e)}"}, status_code=500) - - def handle_pdf_upload(self, session_id: str): - """ - Processes PDF files: extracts text and stores it in the database. - DEPRECATED: This is the old method. Use handle_file_upload instead. - """ - # This function is now deprecated in favor of the new indexing workflow - # but is kept for potential legacy/compatibility reasons. - # For new functionality, it should not be used. - self.send_json_response({ - "warning": "This upload method is deprecated. Use the new file upload and indexing flow.", - "message": "No action taken." - }, status_code=410) # 410 Gone def handle_get_models(self): """Get available models from both Ollama and HuggingFace, grouped by capability""" @@ -762,8 +960,9 @@ def handle_get_models(self): # Add supported HuggingFace embedding models huggingface_embedding_models = [ + "microsoft/harrier-oss-v1-0.6b", # shipped default "Qwen/Qwen3-Embedding-0.6B", - "Qwen/Qwen3-Embedding-4B", + "Qwen/Qwen3-Embedding-4B", "Qwen/Qwen3-Embedding-8B" ] embedding_models.extend(huggingface_embedding_models) @@ -800,9 +999,9 @@ def handle_get_index(self, index_id: str): def handle_create_index(self): try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) + data = self.read_required_json_body() + if data is None: + return name = data.get('name') description = data.get('description') metadata = data.get('metadata', {}) @@ -813,27 +1012,18 @@ def handle_create_index(self): # Add complete metadata from RAG system configuration if available if RAG_SYSTEM_AVAILABLE and PIPELINE_CONFIGS.get('default'): - default_config = PIPELINE_CONFIGS['default'] complete_metadata = { 'status': 'created', 'metadata_source': 'rag_system_config', - 'created_at': json.loads(json.dumps(datetime.now().isoformat())), - 'chunk_size': 512, # From default config - 'chunk_overlap': 64, # From default config - 'retrieval_mode': 'hybrid', # From default config - 'window_size': 5, # From default config - 'embedding_model': 'Qwen/Qwen3-Embedding-0.6B', # From default config - 'enrich_model': 'qwen3:0.6b', # From default config - 'overview_model': 'qwen3:0.6b', # From default config - 'enable_enrich': True, # From default config - 'latechunk': True, # From default config - 'docling_chunk': True, # From default config - 'note': 'Default configuration from RAG system' + 'created_at': datetime.now().isoformat(), + 'note': 'Default configuration from RAG system', } + complete_metadata.update(default_index_metadata()) # Merge with any provided metadata complete_metadata.update(metadata) metadata = complete_metadata - + + idx_id = db.create_index(name, description, metadata) self.send_json_response({'index_id': idx_id}, status_code=201) except Exception as e: @@ -841,27 +1031,30 @@ def handle_create_index(self): def handle_index_file_upload(self, index_id: str): """Reuse file upload logic but store docs under index.""" - form = cgi.FieldStorage(fp=self.rfile, headers=self.headers, environ={'REQUEST_METHOD':'POST', 'CONTENT_TYPE': self.headers['Content-Type']}) uploaded_files=[] - if 'files' in form: - files=form['files'] - if not isinstance(files, list): - files=[files] + incoming = self.parse_multipart_files('files') + if incoming is None: + return # 413 already sent + if incoming: upload_dir='shared_uploads' os.makedirs(upload_dir, exist_ok=True) - for f in files: - if f.filename: - unique=f"{uuid.uuid4()}_{f.filename}" - path=os.path.join(upload_dir, unique) - with open(path,'wb') as out: out.write(f.file.read()) - db.add_document_to_index(index_id, f.filename, os.path.abspath(path)) - uploaded_files.append({'filename':f.filename,'stored_path':os.path.abspath(path)}) + for filename, content in incoming: + unique=f"{uuid.uuid4()}_{filename}" + path=os.path.join(upload_dir, unique) + with open(path,'wb') as out: out.write(content) + db.add_document_to_index(index_id, filename, os.path.abspath(path)) + uploaded_files.append({'filename':filename,'stored_path':os.path.abspath(path)}) if not uploaded_files: self.send_json_response({'error':'No files uploaded'}, status_code=400); return self.send_json_response({'message':f"Uploaded {len(uploaded_files)} files","uploaded_files":uploaded_files}) def handle_build_index(self, index_id: str): try: + # Parse request body for optional flags and configuration. Options the + # caller omits are left out of the payload so the RAG pipeline defaults apply. + # Read before any early return so the request body is always consumed. + options = normalize_options(self.read_json_body(), INDEX_OPTIONS) + index=db.get_index(index_id) if not index: self.send_json_response({'error':'Index not found'}, status_code=404); return @@ -869,95 +1062,27 @@ def handle_build_index(self, index_id: str): if not file_paths: self.send_json_response({'error':'No documents to index'}, status_code=400); return - # Parse request body for optional flags and configuration - latechunk = False - docling_chunk = False - chunk_size = 512 - chunk_overlap = 64 - retrieval_mode = 'hybrid' - window_size = 2 - enable_enrich = True - embedding_model = None - enrich_model = None - batch_size_embed = 50 - batch_size_enrich = 25 - overview_model = None - - if 'Content-Length' in self.headers and int(self.headers['Content-Length']) > 0: - try: - length = int(self.headers['Content-Length']) - body = self.rfile.read(length) - opts = json.loads(body.decode('utf-8')) - latechunk = bool(opts.get('latechunk', False)) - docling_chunk = bool(opts.get('doclingChunk', False)) - chunk_size = int(opts.get('chunkSize', 512)) - chunk_overlap = int(opts.get('chunkOverlap', 64)) - retrieval_mode = str(opts.get('retrievalMode', 'hybrid')) - window_size = int(opts.get('windowSize', 2)) - enable_enrich = bool(opts.get('enableEnrich', True)) - embedding_model = opts.get('embeddingModel') - enrich_model = opts.get('enrichModel') - batch_size_embed = int(opts.get('batchSizeEmbed', 50)) - batch_size_enrich = int(opts.get('batchSizeEnrich', 25)) - overview_model = opts.get('overviewModel') - except Exception: - # Keep defaults on parse error - pass - - # Set per-index overview file path - overview_path = f"index_store/overviews/{index_id}.jsonl" - - # Ensure config_override includes overview_path - def ensure_overview_path(cfg: dict): - cfg["overview_path"] = overview_path - - # we'll inject later when we build config_override - - # Delegate to advanced RAG API same as session indexing - rag_api_url = "http://localhost:8001/index" - import requests, json as _json + # Delegate to the RAG API, same as session indexing + rag_api_url = f"{RAG_API_URL}/index" # Use the index's dedicated LanceDB table so retrieval matches table_name = index.get("vector_table_name") - payload = { + payload: Dict[str, Any] = { "file_paths": file_paths, "session_id": index_id, # reuse index_id for progress tracking "table_name": table_name, - "chunk_size": chunk_size, - "chunk_overlap": chunk_overlap, - "retrieval_mode": retrieval_mode, - "window_size": window_size, - "enable_enrich": enable_enrich, - "batch_size_embed": batch_size_embed, - "batch_size_enrich": batch_size_enrich } - if latechunk: - payload["enable_latechunk"] = True - if docling_chunk: - payload["enable_docling_chunk"] = True - if embedding_model: - payload["embedding_model"] = embedding_model - if enrich_model: - payload["enrich_model"] = enrich_model - if overview_model: - payload["overview_model_name"] = overview_model - - rag_resp = requests.post(rag_api_url, json=payload) + payload.update(options) + + rag_resp = requests.post(rag_api_url, json=payload, timeout=RAG_API_INDEX_TIMEOUT) if rag_resp.status_code==200: - meta_updates = { - "chunk_size": chunk_size, - "chunk_overlap": chunk_overlap, - "retrieval_mode": retrieval_mode, - "window_size": window_size, - "enable_enrich": enable_enrich, - "latechunk": latechunk, - "docling_chunk": docling_chunk, - } - if embedding_model: - meta_updates["embedding_model"] = embedding_model - if enrich_model: - meta_updates["enrich_model"] = enrich_model - if overview_model: - meta_updates["overview_model"] = overview_model + meta_updates: Dict[str, Any] = dict(options) + meta_updates["status"] = "built" + if "enable_latechunk" in meta_updates: + meta_updates["latechunk"] = meta_updates.pop("enable_latechunk") + if "enable_docling_chunk" in meta_updates: + meta_updates["docling_chunk"] = meta_updates.pop("enable_docling_chunk") + if "overview_model_name" in meta_updates: + meta_updates["overview_model"] = meta_updates.pop("overview_model_name") try: db.update_index_metadata(index_id, meta_updates) except Exception as e: @@ -968,23 +1093,21 @@ def ensure_overview_path(cfg: dict): **meta_updates }) else: - # Gracefully handle scenario where table already exists (idempotent build) - try: - err_json = rag_resp.json() - except Exception: - err_json = {} - err_text = err_json.get('error') if isinstance(err_json, dict) else rag_resp.text - if err_text and 'already exists' in err_text: - # Treat as non-fatal; return message indicating index previously built - self.send_json_response({ - "message": "Index already built โ€“ skipping rebuild.", - "note": err_text - }) - else: - self.send_json_response({"error":f"RAG indexing failed: {rag_resp.text}"}, status_code=500) + # The pipeline replaces rows on rebuild (delete-before-add), so + # a non-200 here is always a real failure, never a table-exists + # conflict. + self.send_json_response({"error":f"RAG indexing failed: {rag_resp.text}"}, status_code=500) + except requests.exceptions.Timeout: + self.send_json_response({ + "error": f"Indexing did not complete within {RAG_API_INDEX_TIMEOUT:.0f}s." + }, status_code=504) + except requests.exceptions.ConnectionError: + self.send_json_response({ + "error": f"Could not connect to the RAG API server at {RAG_API_URL}." + }, status_code=502) except Exception as e: self.send_json_response({'error':str(e)}, status_code=500) - + def handle_link_index_to_session(self, session_id: str, index_id: str): try: db.link_index_to_session(session_id, index_id) @@ -1022,6 +1145,50 @@ def handle_delete_index(self, index_id: str): except Exception as e: self.send_json_response({'error': str(e)}, status_code=500) + def handle_save_messages(self, session_id: str): + """Persist a completed streamed turn (the browser streams straight from the + RAG API, so the gateway never sees those messages otherwise).""" + try: + session = db.get_session(session_id) + if not session: + self.send_json_response({"error": "Session not found"}, status_code=404) + return + + content_length = int(self.headers.get('Content-Length', 0)) + if content_length == 0: + self.send_json_response({"error": "Request body required"}, status_code=400) + return + + post_data = self.rfile.read(content_length) + data = json.loads(post_data.decode('utf-8')) + user_message = (data.get('user_message') or '').strip() + assistant_message = (data.get('assistant_message') or '').strip() + source_documents = data.get('source_documents') or [] + steps = data.get('steps') + + if not user_message or not assistant_message: + self.send_json_response({"error": "user_message and assistant_message are required"}, status_code=400) + return + + if session['message_count'] == 0: + db.update_session_title(session_id, generate_session_title(user_message)) + + ai_metadata: Dict[str, Any] = {} + if source_documents: + ai_metadata['source_documents'] = source_documents + if isinstance(steps, list) and steps: + ai_metadata['steps'] = steps + user_message_id = db.add_message(session_id, user_message, "user") + ai_message_id = db.add_message(session_id, assistant_message, "assistant", metadata=ai_metadata or None) + + self.send_json_response({ + "session": db.get_session(session_id), + "user_message_id": user_message_id, + "ai_message_id": ai_message_id, + }) + except Exception as e: + self.send_json_response({"error": str(e)}, status_code=500) + def handle_rename_session(self, session_id: str): """Rename an existing session title""" try: @@ -1064,7 +1231,8 @@ def send_json_response(self, data, status_code: int = 200): self.send_header('Access-Control-Allow-Origin', '*') self.send_header('Access-Control-Allow-Methods', 'GET, POST, PUT, DELETE, OPTIONS') self.send_header('Access-Control-Allow-Headers', 'Content-Type, Authorization') - self.send_header('Access-Control-Allow-Credentials', 'true') + # No Access-Control-Allow-Credentials: the spec forbids combining + # credentialed requests with a '*' origin, and nothing here needs it. self.end_headers() response_bytes = json.dumps(data, indent=2).encode('utf-8') @@ -1081,31 +1249,10 @@ def log_message(self, format, *args): def main(): """Main function to initialize and start the server""" - PORT = 8000 # ๐Ÿ†• Define port try: # Initialize the database print("โœ… Database initialized successfully") - # Initialize the PDF processor - try: - pdf_module.initialize_simple_pdf_processor() - print("๐Ÿ“„ Initializing simple PDF processing...") - if pdf_module.simple_pdf_processor: - print("โœ… Simple PDF processor initialized") - else: - print("โš ๏ธ PDF processing could not be initialized.") - except Exception as e: - print(f"โŒ Error initializing PDF processor: {e}") - print("โš ๏ธ PDF processing disabled - server will run without RAG functionality") - - # Set a global reference to the initialized processor if needed elsewhere - global pdf_processor - pdf_processor = pdf_module.simple_pdf_processor - if pdf_processor: - print("โœ… Global PDF processor initialized") - else: - print("โš ๏ธ PDF processing disabled - server will run without RAG functionality") - # Cleanup empty sessions on startup print("๐Ÿงน Cleaning up empty sessions...") cleanup_count = db.cleanup_empty_sessions() @@ -1119,7 +1266,9 @@ def main(): print(f"๐Ÿš€ Starting localGPT backend server on port {PORT}") print(f"๐Ÿ“ Chat endpoint: http://localhost:{PORT}/chat") print(f"๐Ÿ” Health check: http://localhost:{PORT}/health") - + print(f"๐Ÿง  RAG API: {RAG_API_URL}") + print(f"๐Ÿค– Default generation model: {GENERATION_MODEL}") + # Test Ollama connection client = OllamaClient() if client.is_ollama_running(): diff --git a/backend/simple_pdf_processor.py b/backend/simple_pdf_processor.py deleted file mode 100644 index e1f8dfad..00000000 --- a/backend/simple_pdf_processor.py +++ /dev/null @@ -1,214 +0,0 @@ -""" -Simple PDF Processing Service -Handles PDF upload and text extraction for RAG functionality -""" - -import uuid -from typing import List, Dict, Any -import PyPDF2 -from io import BytesIO -import sqlite3 -from datetime import datetime - -class SimplePDFProcessor: - def __init__(self, db_path: str = "chat_data.db"): - """Initialize simple PDF processor with SQLite storage""" - self.db_path = db_path - self.init_database() - print("โœ… Simple PDF processor initialized") - - def init_database(self): - """Initialize SQLite database for storing PDF content""" - conn = sqlite3.connect(self.db_path) - conn.execute(''' - CREATE TABLE IF NOT EXISTS pdf_documents ( - id TEXT PRIMARY KEY, - session_id TEXT NOT NULL, - filename TEXT NOT NULL, - content TEXT NOT NULL, - created_at TEXT NOT NULL - ) - ''') - - conn.commit() - conn.close() - - def extract_text_from_pdf(self, pdf_bytes: bytes) -> str: - """Extract text from PDF bytes""" - try: - print(f"๐Ÿ“„ Starting PDF text extraction ({len(pdf_bytes)} bytes)") - pdf_file = BytesIO(pdf_bytes) - pdf_reader = PyPDF2.PdfReader(pdf_file) - - print(f"๐Ÿ“– PDF has {len(pdf_reader.pages)} pages") - - text = "" - for page_num, page in enumerate(pdf_reader.pages): - print(f"๐Ÿ“„ Processing page {page_num + 1}") - try: - page_text = page.extract_text() - if page_text.strip(): - text += f"\n--- Page {page_num + 1} ---\n" - text += page_text + "\n" - print(f"โœ… Page {page_num + 1}: extracted {len(page_text)} characters") - except Exception as page_error: - print(f"โŒ Error on page {page_num + 1}: {str(page_error)}") - continue - - print(f"๐Ÿ“„ Total extracted text: {len(text)} characters") - return text.strip() - - except Exception as e: - print(f"โŒ Error extracting text from PDF: {str(e)}") - print(f"โŒ Error type: {type(e).__name__}") - return "" - - def process_pdf(self, pdf_bytes: bytes, filename: str, session_id: str) -> Dict[str, Any]: - """Process a PDF file and store in database""" - print(f"๐Ÿ“„ Processing PDF: {filename}") - - # Extract text - text = self.extract_text_from_pdf(pdf_bytes) - if not text: - return { - "success": False, - "error": "Could not extract text from PDF", - "filename": filename - } - - print(f"๐Ÿ“ Extracted {len(text)} characters from {filename}") - - # Store in database - document_id = str(uuid.uuid4()) - now = datetime.now().isoformat() - - try: - conn = sqlite3.connect(self.db_path) - - # Store document - conn.execute(''' - INSERT INTO pdf_documents (id, session_id, filename, content, created_at) - VALUES (?, ?, ?, ?, ?) - ''', (document_id, session_id, filename, text, now)) - - conn.commit() - conn.close() - - print(f"๐Ÿ’พ Stored document {filename} in database") - - return { - "success": True, - "filename": filename, - "file_id": document_id, - "text_length": len(text) - } - - except Exception as e: - print(f"โŒ Error storing in database: {str(e)}") - return { - "success": False, - "error": f"Database storage failed: {str(e)}", - "filename": filename - } - - def get_session_documents(self, session_id: str) -> List[Dict[str, Any]]: - """Get all documents for a session""" - try: - conn = sqlite3.connect(self.db_path) - conn.row_factory = sqlite3.Row - - cursor = conn.execute(''' - SELECT id, filename, created_at - FROM pdf_documents - WHERE session_id = ? - ORDER BY created_at DESC - ''', (session_id,)) - - documents = [dict(row) for row in cursor.fetchall()] - conn.close() - - return documents - - except Exception as e: - print(f"โŒ Error getting session documents: {str(e)}") - return [] - - def get_document_content(self, session_id: str) -> str: - """Get all document content for a session (for LLM context)""" - try: - conn = sqlite3.connect(self.db_path) - - cursor = conn.execute(''' - SELECT filename, content - FROM pdf_documents - WHERE session_id = ? - ORDER BY created_at ASC - ''', (session_id,)) - - rows = cursor.fetchall() - conn.close() - - if not rows: - return "" - - # Combine all document content - combined_content = "" - for filename, content in rows: - combined_content += f"\n\n=== Document: {filename} ===\n\n" - combined_content += content - - return combined_content.strip() - - except Exception as e: - print(f"โŒ Error getting document content: {str(e)}") - return "" - - def delete_session_documents(self, session_id: str) -> bool: - """Delete all documents for a session""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.execute(''' - DELETE FROM pdf_documents - WHERE session_id = ? - ''', (session_id,)) - - deleted_count = cursor.rowcount - conn.commit() - conn.close() - - if deleted_count > 0: - print(f"๐Ÿ—‘๏ธ Deleted {deleted_count} documents for session {session_id[:8]}...") - - return deleted_count > 0 - - except Exception as e: - print(f"โŒ Error deleting session documents: {str(e)}") - return False - - -# Global instance -simple_pdf_processor = None - -def initialize_simple_pdf_processor(): - """Initialize the global PDF processor""" - global simple_pdf_processor - try: - simple_pdf_processor = SimplePDFProcessor() - print("โœ… Global PDF processor initialized") - except Exception as e: - print(f"โŒ Failed to initialize PDF processor: {str(e)}") - simple_pdf_processor = None - -def get_simple_pdf_processor(): - """Get the global PDF processor instance""" - global simple_pdf_processor - if simple_pdf_processor is None: - initialize_simple_pdf_processor() - return simple_pdf_processor - -if __name__ == "__main__": - # Test the simple PDF processor - print("๐Ÿงช Testing simple PDF processor...") - - processor = SimplePDFProcessor() - print("โœ… Simple PDF processor test completed!") \ No newline at end of file diff --git a/backend/test_backend.py b/backend/test_backend.py deleted file mode 100644 index 0ff94f2f..00000000 --- a/backend/test_backend.py +++ /dev/null @@ -1,153 +0,0 @@ -#!/usr/bin/env python3 -""" -Simple test script for the localGPT backend -""" - -import requests - -def test_health_endpoint(): - """Test the health endpoint""" - print("๐Ÿ” Testing health endpoint...") - try: - response = requests.get("http://localhost:8000/health", timeout=5) - if response.status_code == 200: - data = response.json() - print(f"โœ… Health check passed") - print(f" Ollama running: {data['ollama_running']}") - print(f" Models available: {len(data['available_models'])}") - return True - else: - print(f"โŒ Health check failed: {response.status_code}") - return False - except requests.exceptions.RequestException as e: - print(f"โŒ Health check failed: {e}") - return False - -def test_chat_endpoint(): - """Test the chat endpoint""" - print("\n๐Ÿ’ฌ Testing chat endpoint...") - - test_message = { - "message": "Say 'Hello World' and nothing else.", - "model": "llama3.2:latest" - } - - try: - response = requests.post( - "http://localhost:8000/chat", - headers={"Content-Type": "application/json"}, - json=test_message, - timeout=30 - ) - - if response.status_code == 200: - data = response.json() - print(f"โœ… Chat test passed") - print(f" Model: {data['model']}") - print(f" Response: {data['response']}") - print(f" Message count: {data['message_count']}") - return True - else: - print(f"โŒ Chat test failed: {response.status_code}") - print(f" Response: {response.text}") - return False - - except requests.exceptions.RequestException as e: - print(f"โŒ Chat test failed: {e}") - return False - -def test_conversation_history(): - """Test conversation with history""" - print("\n๐Ÿ—จ๏ธ Testing conversation history...") - - # First message - conversation = [] - - message1 = { - "message": "My name is Alice. Remember this.", - "model": "llama3.2:latest", - "conversation_history": conversation - } - - try: - response1 = requests.post( - "http://localhost:8000/chat", - headers={"Content-Type": "application/json"}, - json=message1, - timeout=30 - ) - - if response1.status_code == 200: - data1 = response1.json() - - # Add to conversation history - conversation.append({"role": "user", "content": "My name is Alice. Remember this."}) - conversation.append({"role": "assistant", "content": data1["response"]}) - - # Second message asking about the name - message2 = { - "message": "What is my name?", - "model": "llama3.2:latest", - "conversation_history": conversation - } - - response2 = requests.post( - "http://localhost:8000/chat", - headers={"Content-Type": "application/json"}, - json=message2, - timeout=30 - ) - - if response2.status_code == 200: - data2 = response2.json() - print(f"โœ… Conversation history test passed") - print(f" First response: {data1['response']}") - print(f" Second response: {data2['response']}") - - # Check if the AI remembered the name - if "alice" in data2['response'].lower(): - print(f"โœ… AI correctly remembered the name!") - else: - print(f"โš ๏ธ AI might not have remembered the name") - return True - else: - print(f"โŒ Second message failed: {response2.status_code}") - return False - else: - print(f"โŒ First message failed: {response1.status_code}") - return False - - except requests.exceptions.RequestException as e: - print(f"โŒ Conversation test failed: {e}") - return False - -def main(): - print("๐Ÿงช Testing localGPT Backend") - print("=" * 40) - - # Test health endpoint - health_ok = test_health_endpoint() - if not health_ok: - print("\nโŒ Backend server is not running or not healthy") - print(" Make sure to run: python server.py") - return - - # Test basic chat - chat_ok = test_chat_endpoint() - if not chat_ok: - print("\nโŒ Chat functionality is not working") - return - - # Test conversation history - conversation_ok = test_conversation_history() - - print("\n" + "=" * 40) - if health_ok and chat_ok and conversation_ok: - print("๐ŸŽ‰ All tests passed! Backend is ready for frontend integration.") - else: - print("โš ๏ธ Some tests failed. Check the issues above.") - - print("\n๐Ÿ”— Ready to connect to frontend at http://localhost:3000") - -if __name__ == "__main__": - main() \ No newline at end of file diff --git a/backend/test_gateway_routing.py b/backend/test_gateway_routing.py new file mode 100644 index 00000000..c3b1acc8 --- /dev/null +++ b/backend/test_gateway_routing.py @@ -0,0 +1,164 @@ +"""Unit tests for the gateway's deterministic routing gate (no HTTP, no LLM). + +Covers `should_use_rag` / `is_smalltalk_or_meta` in `backend/server.py`, the +retrieval-first cascade that replaced the per-message enrichment-model router: + + force_rag โ†’ RAG ยท no indexes โ†’ direct ยท smalltalk/meta โ†’ direct ยท else RAG + +Run it: + + .venv/bin/python backend/test_gateway_routing.py +""" + +import os +import sys + +BACKEND_DIR = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, BACKEND_DIR) + +from server import is_smalltalk_or_meta, should_use_rag # noqa: E402 + +IDX = ["2fb7a91a-e7ae-46f5-93cd-c8ac52b3ea79"] +NO_IDX: list = [] + +# (message, expected_use_rag, why) +CASES = [ + # --- planted-fact questions from eval/smoke_e2e.py must reach retrieval --- + ("What pressure does the brew boiler operate at during extraction?", True, "planted fact"), + ("Which sensor part should be replaced when error code E11 appears?", True, "planted fact"), + ("How long is the Atlas-7 parts warranty?", True, "planted fact"), + ("Where is the serial number engraved?", True, "planted fact"), + ("Summarize the service manual.", True, "document operation"), + ("Who manufactures the Atlas-7?", True, "entity question"), + ("9.2 bar?", True, "terse but factual"), + ("descaling", True, "bare keyword, no core smalltalk phrase"), + + # --- the old defect: any message containing "test" went direct --- + ("Which test procedure applies after replacing the pump?", True, "old defect: 'test'"), + ("What is the test point voltage on the control board?", True, "old defect: 'test'"), + ("Explain the E11 diagnostic test in detail.", True, "old defect: 'test'"), + ("test", True, "bare 'test' is not an allowlisted smalltalk phrase"), + ("Is this document checked and verified?", True, "old defect: 'check'"), + ("How do I test the group head gasket?", True, "old defect: 'test'"), + + # --- smalltalk shortcuts --- + ("hello", False, "greeting"), + ("Hello!", False, "greeting, punctuated"), + ("hi there", False, "greeting + filler"), + ("hey", False, "greeting"), + ("Good morning", False, "greeting"), + ("how are you?", False, "greeting"), + ("thanks!", False, "thanks"), + ("Thank you so much", False, "thanks"), + ("thx", False, "thanks"), + ("bye", False, "farewell"), + ("goodbye, see you later", False, "farewell"), + ("ok", False, "acknowledgement"), + ("got it, thanks", False, "acknowledgement + thanks"), + ("nevermind", False, "acknowledgement"), + ("", False, "empty message never needs retrieval"), + (" ", False, "whitespace-only"), + + # --- assistant-meta shortcuts --- + ("who are you?", False, "meta"), + ("Who are you", False, "meta"), + ("what model are you", False, "meta"), + ("what model are you?", False, "meta"), + ("which model do you use?", False, "meta"), + ("what are you?", False, "meta"), + ("what is your name?", False, "meta"), + ("are you an AI?", False, "meta"), + ("who built you?", False, "meta"), + ("what can you do?", False, "meta"), + ("tell me about yourself", False, "meta"), + + # --- smalltalk words inside a real question must NOT shortcut --- + ("Hello, what is the brew boiler pressure?", True, "greeting prefix + real question"), + ("Thanks - now summarize section 4 for me", True, "thanks prefix + real question"), + ("Who are the authors of the service manual?", True, "'who are' but not about the assistant"), + ("What model number is the pressure sensor?", True, "'what model' but not about the assistant"), + ("Is the machine ok to run at 1.45 bar?", True, "'ok' inside a real question"), + ("no problem code is listed for E12?", True, "'no problem' inside a real question"), + ("What is a good night mode setting?", True, "'good night' inside a real question"), +] + + +def check(label, actual, expected, why, kind="route"): + ok = actual == expected + fmt = (lambda v: ("RAG" if v else "DIRECT")) if kind == "route" else (lambda v: str(v)) + print(f" {'PASS' if ok else 'FAIL'} {label:<62} -> {fmt(actual):<6}" + f" (expected {fmt(expected)}; {why})") + return ok + + +def main(): + failures = 0 + total = 0 + + print("\n[1] Session WITH linked indexes") + for message, expected, why in CASES: + total += 1 + label = repr(message) if len(message) < 60 else repr(message[:57] + "...") + if not check(label, should_use_rag(message, IDX), expected, why): + failures += 1 + + print("\n[2] Session with NO linked indexes -> always direct") + for message, _expected, _why in CASES: + total += 1 + label = repr(message) if len(message) < 60 else repr(message[:57] + "...") + if not check(label, should_use_rag(message, NO_IDX), False, "no indexes linked"): + failures += 1 + for empty in (None, [], ()): + total += 1 + if not check(f"idx_ids={empty!r}", should_use_rag("What is the brew pressure?", empty), + False, "no indexes linked"): + failures += 1 + + print("\n[3] force_rag is honored") + force_cases = [ + ("hello", IDX), + ("thanks!", IDX), + ("who are you?", IDX), + ("What pressure does the brew boiler operate at?", IDX), + ("hello", NO_IDX), # force_rag wins even with no indexes + ("", NO_IDX), + ] + for message, idx in force_cases: + total += 1 + if not check(f"force_rag {message!r} idx={bool(idx)}", + should_use_rag(message, idx, force_rag=True), True, "force_rag=True"): + failures += 1 + + print("\n[4] No LLM call is made by the gate") + # server.should_use_rag must not reference an Ollama client at all: the gate + # is a module-level function with no client argument and no network use. + import inspect + import server + total += 1 + src = inspect.getsource(server.should_use_rag) + inspect.getsource(server.is_smalltalk_or_meta) + clean = not any(tok in src for tok in ("ollama", "requests", "ENRICHMENT_MODEL", "http")) + if not check("gate source is free of LLM/network calls", clean, True, + "deterministic gate", kind="bool"): + failures += 1 + total += 1 + removed = not any(hasattr(server.ChatHandler, name) for name in + ("_should_use_rag", "_simple_pattern_routing", + "_route_using_overviews", "_load_document_overviews")) + if not check("old LLM router + pattern fallback deleted", removed, True, + "no _simple_pattern_routing / _route_using_overviews", kind="bool"): + failures += 1 + + print("\n[5] is_smalltalk_or_meta is index-independent") + for message, expected_rag, why in CASES: + total += 1 + # With indexes linked, should_use_rag is exactly `not is_smalltalk_or_meta`. + if not check(f"consistency {message[:40]!r}", + should_use_rag(message, IDX), not is_smalltalk_or_meta(message), why): + failures += 1 + + print(f"\n{total - failures}/{total} checks passed") + return 1 if failures else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/backend/test_ollama_connectivity.py b/backend/test_ollama_connectivity.py deleted file mode 100644 index d4e2e65c..00000000 --- a/backend/test_ollama_connectivity.py +++ /dev/null @@ -1,37 +0,0 @@ -#!/usr/bin/env python3 - -import os -import sys - -def test_ollama_connectivity(): - """Test Ollama connectivity from within Docker container""" - print("๐Ÿงช Testing Ollama Connectivity") - print("=" * 40) - - ollama_host = os.getenv('OLLAMA_HOST', 'Not set') - print(f"OLLAMA_HOST environment variable: {ollama_host}") - - try: - from ollama_client import OllamaClient - client = OllamaClient() - print(f"OllamaClient base_url: {client.base_url}") - - is_running = client.is_ollama_running() - print(f"Ollama running: {is_running}") - - if is_running: - models = client.list_models() - print(f"Available models: {models}") - print("โœ… Ollama connectivity test passed!") - return True - else: - print("โŒ Ollama connectivity test failed!") - return False - - except Exception as e: - print(f"โŒ Error testing Ollama connectivity: {e}") - return False - -if __name__ == "__main__": - success = test_ollama_connectivity() - sys.exit(0 if success else 1) diff --git a/batch_indexing_config.json b/batch_indexing_config.json deleted file mode 100644 index 8ac66256..00000000 --- a/batch_indexing_config.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "index_name": "Sample Batch Index", - "index_description": "Example batch index configuration", - "documents": [ - "./rag_system/documents/invoice_1039.pdf", - "./rag_system/documents/invoice_1041.pdf" - ], - "processing": { - "chunk_size": 512, - "chunk_overlap": 64, - "enable_enrich": true, - "enable_latechunk": true, - "enable_docling": true, - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", - "generation_model": "qwen3:0.6b", - "retrieval_mode": "hybrid", - "window_size": 2 - } -} \ No newline at end of file diff --git a/create_index_script.py b/create_index_script.py index dc7b894e..dadd17f2 100644 --- a/create_index_script.py +++ b/create_index_script.py @@ -8,10 +8,12 @@ Usage: python create_index_script.py - python create_index_script.py --batch - python create_index_script.py --config custom_config.json + python create_index_script.py --batch index_config.json + python create_index_script.py --config custom_pipeline_config.json + python create_index_script.py --create-sample """ +import copy import os import sys import json @@ -23,7 +25,8 @@ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) try: - from rag_system.main import PIPELINE_CONFIGS, get_agent + from rag_system.main import PIPELINE_CONFIGS, OLLAMA_CONFIG + from rag_system.factory import get_agent from rag_system.pipelines.indexing_pipeline import IndexingPipeline from rag_system.utils.ollama_client import OllamaClient from backend.database import ChatDatabase @@ -35,26 +38,14 @@ class IndexCreator: """Interactive index creation utility.""" - + def __init__(self, config_path: Optional[str] = None): """Initialize the index creator with optional custom configuration.""" self.db = ChatDatabase() - self.config = self._load_config(config_path) - - # Initialize Ollama client - self.ollama_client = OllamaClient() - self.ollama_config = { - "generation_model": "qwen3:0.6b", - "embedding_model": "qwen3:0.6b" - } - - # Initialize indexing pipeline - self.pipeline = IndexingPipeline( - self.config, - self.ollama_client, - self.ollama_config - ) - + self.base_config = self._load_config(config_path) + self.ollama_config = OLLAMA_CONFIG + self.ollama_client = OllamaClient(host=OLLAMA_CONFIG["host"]) + def _load_config(self, config_path: Optional[str] = None) -> dict: """Load configuration from file or use default.""" if config_path and os.path.exists(config_path): @@ -64,9 +55,37 @@ def _load_config(self, config_path: Optional[str] = None) -> dict: except Exception as e: print(f"โš ๏ธ Error loading config from {config_path}: {e}") print("Using default configuration...") - + return PIPELINE_CONFIGS.get("default", {}) - + + def _build_pipeline(self, index_id: str, processing: dict) -> IndexingPipeline: + """Build a pipeline that writes into this index's own tables.""" + config = copy.deepcopy(self.base_config) + table_name = f"text_pages_{index_id}" + + config.setdefault("storage", {})["text_table_name"] = table_name + # The pipeline reads whichever of these two keys is present. + retrievers = config.get("retrievers") + if retrievers is None: + retrievers = config.setdefault("retrieval", {}) + retrievers.setdefault("dense", {})["lancedb_table_name"] = table_name + retrievers.setdefault("latechunk", {})["enabled"] = bool(processing.get("enable_latechunk", False)) + + config["chunker_mode"] = "docling" if processing.get("enable_docling", True) else "legacy" + config.setdefault("chunking", {})["chunk_size"] = int(processing.get("chunk_size", 512)) + config.setdefault("contextual_enricher", {}).update({ + "enabled": bool(processing.get("enable_enrich", True)), + "window_size": int(processing.get("window_size", 2)), + }) + config["overview_path"] = f"index_store/overviews/{index_id}.jsonl" + + if processing.get("embedding_model"): + config["embedding_model_name"] = processing["embedding_model"] + if processing.get("enrich_model"): + config["enrich_model"] = processing["enrich_model"] + + return IndexingPipeline(config, self.ollama_client, self.ollama_config) + def get_user_input(self, prompt: str, default: str = "") -> str: """Get user input with optional default value.""" if default: @@ -148,33 +167,34 @@ def configure_processing(self) -> dict: print("Configure how documents will be processed:") # Basic settings - chunk_size = int(self.get_user_input("Chunk size", "512")) - chunk_overlap = int(self.get_user_input("Chunk overlap", "64")) - + chunk_size = int(self.get_user_input("Chunk size (tokens)", "512")) + # Advanced settings print("\nAdvanced options:") enable_enrich = self.get_user_input("Enable contextual enrichment? (y/n)", "y").lower() == 'y' enable_latechunk = self.get_user_input("Enable late chunking? (y/n)", "y").lower() == 'y' enable_docling = self.get_user_input("Enable Docling chunking? (y/n)", "y").lower() == 'y' - + # Model selection print("\nModel Configuration:") - embedding_model = self.get_user_input("Embedding model", "Qwen/Qwen3-Embedding-0.6B") - generation_model = self.get_user_input("Generation model", "qwen3:0.6b") - + default_embedding = self.base_config.get("embedding_model_name", "") + embedding_model = self.get_user_input("Embedding model", default_embedding) + enrich_model = self.get_user_input( + "Enrichment model", self.ollama_config.get("enrichment_model", "") + ) + return { "chunk_size": chunk_size, - "chunk_overlap": chunk_overlap, "enable_enrich": enable_enrich, "enable_latechunk": enable_latechunk, "enable_docling": enable_docling, "embedding_model": embedding_model, - "generation_model": generation_model, + "enrich_model": enrich_model, "retrieval_mode": "hybrid", "window_size": 2 } - - def create_index_interactive(self) -> None: + + def create_index_interactive(self) -> bool: """Run the interactive index creation process.""" print("๐Ÿš€ LocalGPT Index Creation Tool") print("=" * 50) @@ -201,41 +221,48 @@ def create_index_interactive(self) -> None: if self.get_user_input("\nProceed with index creation? (y/n)", "y").lower() != 'y': print("โŒ Index creation cancelled.") - return - + return False + # Create the index + index_id = None try: print("\n๐Ÿ”ฅ Creating index...") - + # Create index record in database index_id = self.db.create_index( name=index_name, description=index_description, metadata=processing_config ) - + # Add documents to index for doc_path in documents: filename = os.path.basename(doc_path) self.db.add_document_to_index(index_id, filename, doc_path) - + # Process documents through pipeline print("๐Ÿ“š Processing documents...") - self.pipeline.process_documents(documents) - + self._build_pipeline(index_id, processing_config).run(documents) + print(f"\nโœ… Index '{index_name}' created successfully!") print(f"Index ID: {index_id}") print(f"Processed {len(documents)} documents") - + # Test the index if self.get_user_input("\nTest the index with a sample query? (y/n)", "y").lower() == 'y': self.test_index(index_id) - + + return True + except Exception as e: print(f"โŒ Error creating index: {e}") import traceback traceback.print_exc() - + if index_id: + print(f"๐Ÿงน Removing incomplete index record {index_id}") + self.db.delete_index(index_id) + return False + def test_index(self, index_id: str) -> None: """Test the created index with a sample query.""" try: @@ -257,58 +284,67 @@ def test_index(self, index_id: str) -> None: except Exception as e: print(f"โŒ Error testing index: {e}") - def batch_create_from_config(self, config_file: str) -> None: + def batch_create_from_config(self, config_file: str) -> bool: """Create index from batch configuration file.""" + index_id = None try: with open(config_file, 'r') as f: batch_config = json.load(f) - + index_name = batch_config.get("index_name", "Batch Index") index_description = batch_config.get("index_description", "") documents = batch_config.get("documents", []) processing_config = batch_config.get("processing", {}) - + if not documents: print("โŒ No documents specified in batch configuration") - return - + return False + # Validate documents exist valid_documents = [] for doc_path in documents: if os.path.exists(doc_path): - valid_documents.append(doc_path) + valid_documents.append(os.path.abspath(doc_path)) else: print(f"โš ๏ธ Document not found: {doc_path}") - + if not valid_documents: print("โŒ No valid documents found") - return - + return False + print(f"๐Ÿš€ Creating batch index: {index_name}") print(f"๐Ÿ“„ Processing {len(valid_documents)} documents...") - + # Create index index_id = self.db.create_index( name=index_name, description=index_description, metadata=processing_config ) - + # Add documents for doc_path in valid_documents: filename = os.path.basename(doc_path) self.db.add_document_to_index(index_id, filename, doc_path) - + # Process documents - self.pipeline.process_documents(valid_documents) - + self._build_pipeline(index_id, processing_config).run(valid_documents) + print(f"โœ… Batch index '{index_name}' created successfully!") print(f"Index ID: {index_id}") - + return True + except Exception as e: print(f"โŒ Error creating batch index: {e}") import traceback traceback.print_exc() + if index_id: + print(f"๐Ÿงน Removing incomplete index record {index_id}") + self.db.delete_index(index_id) + return False + + +SAMPLE_CONFIG_FILENAME = "index_config.sample.json" def create_sample_batch_config(): @@ -317,56 +353,59 @@ def create_sample_batch_config(): "index_name": "Sample Batch Index", "index_description": "Example batch index configuration", "documents": [ - "./rag_system/documents/invoice_1039.pdf", - "./rag_system/documents/invoice_1041.pdf" + "/absolute/path/to/first.pdf", + "/absolute/path/to/second.pdf" ], "processing": { "chunk_size": 512, - "chunk_overlap": 64, "enable_enrich": True, "enable_latechunk": True, "enable_docling": True, - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", - "generation_model": "qwen3:0.6b", + "embedding_model": PIPELINE_CONFIGS["default"]["embedding_model_name"], + "enrich_model": OLLAMA_CONFIG["enrichment_model"], "retrieval_mode": "hybrid", "window_size": 2 } } - - with open("batch_indexing_config.json", "w") as f: + + with open(SAMPLE_CONFIG_FILENAME, "w") as f: json.dump(sample_config, f, indent=2) - - print("๐Ÿ“„ Sample batch configuration created: batch_indexing_config.json") + + print(f"๐Ÿ“„ Sample batch configuration created: {SAMPLE_CONFIG_FILENAME}") -def main(): +def main() -> int: """Main entry point for the script.""" parser = argparse.ArgumentParser(description="LocalGPT Index Creation Tool") parser.add_argument("--batch", help="Batch configuration file", type=str) parser.add_argument("--config", help="Custom pipeline configuration file", type=str) parser.add_argument("--create-sample", action="store_true", help="Create sample batch config") - + args = parser.parse_args() - + if args.create_sample: create_sample_batch_config() - return - + return 0 + try: creator = IndexCreator(config_path=args.config) - + if args.batch: - creator.batch_create_from_config(args.batch) + ok = creator.batch_create_from_config(args.batch) else: - creator.create_index_interactive() - + ok = creator.create_index_interactive() + + return 0 if ok else 1 + except KeyboardInterrupt: print("\n\nโŒ Operation cancelled by user.") + return 130 except Exception as e: print(f"โŒ Unexpected error: {e}") import traceback traceback.print_exc() + return 1 if __name__ == "__main__": - main() \ No newline at end of file + sys.exit(main()) diff --git a/demo_batch_indexing.py b/demo_batch_indexing.py deleted file mode 100644 index 06a1847b..00000000 --- a/demo_batch_indexing.py +++ /dev/null @@ -1,386 +0,0 @@ -#!/usr/bin/env python3 -""" -Demo Batch Indexing Script for LocalGPT RAG System - -This script demonstrates how to perform batch indexing of multiple documents -using configuration files. It's designed to showcase the full capabilities -of the indexing pipeline with various configuration options. - -Usage: - python demo_batch_indexing.py --config batch_indexing_config.json - python demo_batch_indexing.py --create-sample-config - python demo_batch_indexing.py --help -""" - -import os -import sys -import json -import argparse -import time -import logging -from typing import List, Dict, Any, Optional -from pathlib import Path -from datetime import datetime - -# Add the project root to the path so we can import rag_system modules -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) - -try: - from rag_system.main import PIPELINE_CONFIGS - from rag_system.pipelines.indexing_pipeline import IndexingPipeline - from rag_system.utils.ollama_client import OllamaClient - from backend.database import ChatDatabase -except ImportError as e: - print(f"โŒ Error importing required modules: {e}") - print("Please ensure you're running this script from the project root directory.") - sys.exit(1) - -# Configure logging -logging.basicConfig( - level=logging.INFO, - format="%(asctime)s | %(levelname)-7s | %(name)s | %(message)s", -) - - -class BatchIndexingDemo: - """Demonstration of batch indexing capabilities.""" - - def __init__(self, config_path: str): - """Initialize the batch indexing demo.""" - self.config_path = config_path - self.config = self._load_config() - self.db = ChatDatabase() - - # Initialize Ollama client - self.ollama_client = OllamaClient() - - # Initialize pipeline with merged configuration - self.pipeline_config = self._merge_configurations() - self.pipeline = IndexingPipeline( - self.pipeline_config, - self.ollama_client, - self.config.get("ollama_config", { - "generation_model": "qwen3:0.6b", - "embedding_model": "qwen3:0.6b" - }) - ) - - def _load_config(self) -> Dict[str, Any]: - """Load batch indexing configuration from file.""" - try: - with open(self.config_path, 'r') as f: - config = json.load(f) - print(f"โœ… Loaded configuration from {self.config_path}") - return config - except FileNotFoundError: - print(f"โŒ Configuration file not found: {self.config_path}") - sys.exit(1) - except json.JSONDecodeError as e: - print(f"โŒ Invalid JSON in configuration file: {e}") - sys.exit(1) - - def _merge_configurations(self) -> Dict[str, Any]: - """Merge batch config with default pipeline config.""" - # Start with default pipeline configuration - merged_config = PIPELINE_CONFIGS.get("default", {}).copy() - - # Override with batch-specific settings - batch_settings = self.config.get("pipeline_settings", {}) - - # Deep merge for nested dictionaries - def deep_merge(base: dict, override: dict) -> dict: - result = base.copy() - for key, value in override.items(): - if key in result and isinstance(result[key], dict) and isinstance(value, dict): - result[key] = deep_merge(result[key], value) - else: - result[key] = value - return result - - return deep_merge(merged_config, batch_settings) - - def validate_documents(self, documents: List[str]) -> List[str]: - """Validate and filter document paths.""" - valid_documents = [] - - print(f"๐Ÿ“‹ Validating {len(documents)} documents...") - - for doc_path in documents: - # Handle relative paths - if not os.path.isabs(doc_path): - doc_path = os.path.abspath(doc_path) - - if os.path.exists(doc_path): - # Check file extension - ext = Path(doc_path).suffix.lower() - if ext in ['.pdf', '.txt', '.docx', '.md', '.html', '.htm']: - valid_documents.append(doc_path) - print(f" โœ… {doc_path}") - else: - print(f" โš ๏ธ Unsupported file type: {doc_path}") - else: - print(f" โŒ File not found: {doc_path}") - - print(f"๐Ÿ“Š {len(valid_documents)} valid documents found") - return valid_documents - - def create_indexes(self) -> List[str]: - """Create multiple indexes based on configuration.""" - indexes = self.config.get("indexes", []) - created_indexes = [] - - for index_config in indexes: - index_id = self.create_single_index(index_config) - if index_id: - created_indexes.append(index_id) - - return created_indexes - - def create_single_index(self, index_config: Dict[str, Any]) -> Optional[str]: - """Create a single index from configuration.""" - try: - # Extract index metadata - index_name = index_config.get("name", "Unnamed Index") - index_description = index_config.get("description", "") - documents = index_config.get("documents", []) - - if not documents: - print(f"โš ๏ธ No documents specified for index '{index_name}', skipping...") - return None - - # Validate documents - valid_documents = self.validate_documents(documents) - if not valid_documents: - print(f"โŒ No valid documents found for index '{index_name}'") - return None - - print(f"\n๐Ÿš€ Creating index: {index_name}") - print(f"๐Ÿ“„ Processing {len(valid_documents)} documents") - - # Create index record in database - index_metadata = { - "created_by": "demo_batch_indexing.py", - "created_at": datetime.now().isoformat(), - "document_count": len(valid_documents), - "config_used": index_config.get("processing_options", {}) - } - - index_id = self.db.create_index( - name=index_name, - description=index_description, - metadata=index_metadata - ) - - # Add documents to index - for doc_path in valid_documents: - filename = os.path.basename(doc_path) - self.db.add_document_to_index(index_id, filename, doc_path) - - # Process documents through pipeline - start_time = time.time() - self.pipeline.process_documents(valid_documents) - processing_time = time.time() - start_time - - print(f"โœ… Index '{index_name}' created successfully!") - print(f" Index ID: {index_id}") - print(f" Processing time: {processing_time:.2f} seconds") - print(f" Documents processed: {len(valid_documents)}") - - return index_id - - except Exception as e: - print(f"โŒ Error creating index '{index_name}': {e}") - import traceback - traceback.print_exc() - return None - - def demonstrate_features(self): - """Demonstrate various indexing features.""" - print("\n๐ŸŽฏ Batch Indexing Demo Features:") - print("=" * 50) - - # Show configuration - print(f"๐Ÿ“‹ Configuration file: {self.config_path}") - print(f"๐Ÿ“Š Number of indexes to create: {len(self.config.get('indexes', []))}") - - # Show pipeline settings - pipeline_settings = self.config.get("pipeline_settings", {}) - if pipeline_settings: - print("\nโš™๏ธ Pipeline Settings:") - for key, value in pipeline_settings.items(): - print(f" {key}: {value}") - - # Show model configuration - ollama_config = self.config.get("ollama_config", {}) - if ollama_config: - print("\n๐Ÿค– Model Configuration:") - for key, value in ollama_config.items(): - print(f" {key}: {value}") - - def run_demo(self): - """Run the complete batch indexing demo.""" - print("๐Ÿš€ LocalGPT Batch Indexing Demo") - print("=" * 50) - - # Show demo features - self.demonstrate_features() - - # Create indexes - print(f"\n๐Ÿ“š Starting batch indexing process...") - start_time = time.time() - - created_indexes = self.create_indexes() - - total_time = time.time() - start_time - - # Summary - print(f"\n๐Ÿ“Š Batch Indexing Summary") - print("=" * 50) - print(f"โœ… Successfully created {len(created_indexes)} indexes") - print(f"โฑ๏ธ Total processing time: {total_time:.2f} seconds") - - if created_indexes: - print(f"\n๐Ÿ“‹ Created Indexes:") - for i, index_id in enumerate(created_indexes, 1): - index_info = self.db.get_index(index_id) - if index_info: - print(f" {i}. {index_info['name']} ({index_id[:8]}...)") - print(f" Documents: {len(index_info.get('documents', []))}") - - print(f"\n๐ŸŽ‰ Demo completed successfully!") - print(f"๐Ÿ’ก You can now use these indexes in the LocalGPT interface.") - - -def create_sample_config(): - """Create a comprehensive sample configuration file.""" - sample_config = { - "description": "Demo batch indexing configuration showcasing various features", - "pipeline_settings": { - "embedding_model_name": "Qwen/Qwen3-Embedding-0.6B", - "indexing": { - "embedding_batch_size": 50, - "enrichment_batch_size": 25, - "enable_progress_tracking": True - }, - "contextual_enricher": { - "enabled": True, - "window_size": 2, - "model_name": "qwen3:0.6b" - }, - "chunking": { - "chunk_size": 512, - "chunk_overlap": 64, - "enable_latechunk": True, - "enable_docling": True - }, - "retrievers": { - "dense": { - "enabled": True, - "lancedb_table_name": "demo_text_pages" - }, - "bm25": { - "enabled": True, - "index_name": "demo_bm25_index" - } - }, - "storage": { - "lancedb_uri": "./index_store/lancedb", - "bm25_path": "./index_store/bm25" - } - }, - "ollama_config": { - "generation_model": "qwen3:0.6b", - "embedding_model": "qwen3:0.6b" - }, - "indexes": [ - { - "name": "Sample Invoice Collection", - "description": "Demo index containing sample invoice documents", - "documents": [ - "./rag_system/documents/invoice_1039.pdf", - "./rag_system/documents/invoice_1041.pdf" - ], - "processing_options": { - "chunk_size": 512, - "enable_enrichment": True, - "retrieval_mode": "hybrid" - } - }, - { - "name": "Research Papers Demo", - "description": "Demo index for research papers and whitepapers", - "documents": [ - "./rag_system/documents/Newwhitepaper_Agents2.pdf" - ], - "processing_options": { - "chunk_size": 1024, - "enable_enrichment": True, - "retrieval_mode": "dense" - } - } - ] - } - - config_filename = "batch_indexing_config.json" - with open(config_filename, "w") as f: - json.dump(sample_config, f, indent=2) - - print(f"โœ… Sample configuration created: {config_filename}") - print(f"๐Ÿ“ Edit this file to customize your batch indexing setup") - print(f"๐Ÿš€ Run: python demo_batch_indexing.py --config {config_filename}") - - -def main(): - """Main entry point for the demo script.""" - parser = argparse.ArgumentParser( - description="LocalGPT Batch Indexing Demo", - formatter_class=argparse.RawDescriptionHelpFormatter, - epilog=""" -Examples: - python demo_batch_indexing.py --config batch_indexing_config.json - python demo_batch_indexing.py --create-sample-config - -This demo showcases the advanced batch indexing capabilities of LocalGPT, -including multi-index creation, advanced configuration options, and -comprehensive processing pipelines. - """ - ) - - parser.add_argument( - "--config", - type=str, - default="batch_indexing_config.json", - help="Path to batch indexing configuration file" - ) - - parser.add_argument( - "--create-sample-config", - action="store_true", - help="Create a sample configuration file" - ) - - args = parser.parse_args() - - if args.create_sample_config: - create_sample_config() - return - - if not os.path.exists(args.config): - print(f"โŒ Configuration file not found: {args.config}") - print(f"๐Ÿ’ก Create a sample config with: python {sys.argv[0]} --create-sample-config") - sys.exit(1) - - try: - demo = BatchIndexingDemo(args.config) - demo.run_demo() - - except KeyboardInterrupt: - print("\n\nโŒ Demo cancelled by user.") - except Exception as e: - print(f"โŒ Demo failed: {e}") - import traceback - traceback.print_exc() - - -if __name__ == "__main__": - main() \ No newline at end of file diff --git a/docker-compose.local-ollama.yml b/docker-compose.local-ollama.yml deleted file mode 100644 index a51dee0a..00000000 --- a/docker-compose.local-ollama.yml +++ /dev/null @@ -1,77 +0,0 @@ -services: - # RAG API server (connects to host Ollama) - rag-api: - build: - context: . - dockerfile: Dockerfile.rag-api - container_name: rag-api - ports: - - "8001:8001" - environment: - - OLLAMA_HOST=http://host.docker.internal:11434 - - NODE_ENV=production - volumes: - - ./lancedb:/app/lancedb - - ./index_store:/app/index_store - - ./shared_uploads:/app/shared_uploads - healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:8001/models"] - interval: 30s - timeout: 10s - retries: 3 - restart: unless-stopped - networks: - - rag-network - - # Backend API server - backend: - build: - context: . - dockerfile: Dockerfile.backend - container_name: rag-backend - ports: - - "8000:8000" - environment: - - NODE_ENV=production - - RAG_API_URL=http://rag-api:8001 - volumes: - - ./backend/chat_data.db:/app/backend/chat_data.db - - ./shared_uploads:/app/shared_uploads - depends_on: - rag-api: - condition: service_healthy - healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:8000/health"] - interval: 30s - timeout: 10s - retries: 3 - restart: unless-stopped - networks: - - rag-network - - # Frontend Next.js application - frontend: - build: - context: . - dockerfile: Dockerfile.frontend - container_name: rag-frontend - ports: - - "3000:3000" - environment: - - NODE_ENV=production - - NEXT_PUBLIC_API_URL=http://localhost:8000 - depends_on: - backend: - condition: service_healthy - healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:3000"] - interval: 30s - timeout: 10s - retries: 3 - restart: unless-stopped - networks: - - rag-network - -networks: - rag-network: - driver: bridge \ No newline at end of file diff --git a/docker-compose.yml b/docker-compose.yml index abf8b80f..394c01d0 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -32,12 +32,22 @@ services: # Use host Ollama by default, or containerized Ollama if enabled - OLLAMA_HOST=${OLLAMA_HOST:-http://host.docker.internal:11434} - NODE_ENV=production + - DB_PATH=${DB_PATH:-/app/backend/chat_data.db} + - LANCEDB_PATH=${LANCEDB_PATH:-/app/lancedb} + - GENERATION_MODEL=${GENERATION_MODEL:-qwen3.5:9b} + - ENRICHMENT_MODEL=${ENRICHMENT_MODEL:-qwen3.5:4b} + - EMBEDDING_MODEL=${EMBEDDING_MODEL:-microsoft/harrier-oss-v1-0.6b} + - RERANKER_MODEL=${RERANKER_MODEL:-Qwen/Qwen3-Reranker-4B} + extra_hosts: + - "host.docker.internal:host-gateway" volumes: - ./lancedb:/app/lancedb - ./index_store:/app/index_store - ./shared_uploads:/app/shared_uploads + # Shared SQLite database - same mount as the backend service + - ./backend:/app/backend healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:8001/models"] + test: ["CMD", "curl", "-f", "http://localhost:8001/health"] interval: 30s timeout: 10s retries: 3 @@ -55,11 +65,20 @@ services: - "8000:8000" environment: - NODE_ENV=production - - RAG_API_URL=http://rag-api:8001 - - OLLAMA_HOST=${OLLAMA_HOST:-http://172.18.0.1:11434} + - RAG_API_URL=${RAG_API_URL:-http://rag-api:8001} + - OLLAMA_HOST=${OLLAMA_HOST:-http://host.docker.internal:11434} + - DB_PATH=${DB_PATH:-/app/backend/chat_data.db} + - LANCEDB_PATH=${LANCEDB_PATH:-/app/lancedb} + - GENERATION_MODEL=${GENERATION_MODEL:-qwen3.5:9b} + - ENRICHMENT_MODEL=${ENRICHMENT_MODEL:-qwen3.5:4b} + extra_hosts: + - "host.docker.internal:host-gateway" volumes: + # Shared SQLite database - same mount as the rag-api service - ./backend:/app/backend - ./shared_uploads:/app/shared_uploads + - ./index_store:/app/index_store + - ./lancedb:/app/lancedb depends_on: rag-api: condition: service_healthy @@ -77,17 +96,24 @@ services: build: context: . dockerfile: Dockerfile.frontend + args: + NEXT_PUBLIC_API_URL: ${NEXT_PUBLIC_API_URL:-http://localhost:8000} + NEXT_PUBLIC_RAG_API_URL: ${NEXT_PUBLIC_RAG_API_URL:-http://localhost:8001} container_name: rag-frontend ports: - "3000:3000" environment: - NODE_ENV=production - - NEXT_PUBLIC_API_URL=http://localhost:8000 + - NEXT_PUBLIC_API_URL=${NEXT_PUBLIC_API_URL:-http://localhost:8000} + - NEXT_PUBLIC_RAG_API_URL=${NEXT_PUBLIC_RAG_API_URL:-http://localhost:8001} + extra_hosts: + - "host.docker.internal:host-gateway" depends_on: backend: condition: service_healthy healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:3000"] + # node:20-alpine ships busybox wget, not curl + test: ["CMD-SHELL", "wget -qO- http://localhost:3000 >/dev/null 2>&1 || exit 1"] interval: 30s timeout: 10s retries: 3 @@ -101,4 +127,4 @@ volumes: networks: rag-network: - driver: bridge \ No newline at end of file + driver: bridge diff --git a/docker.env b/docker.env index 4bb29eed..6b9a57f6 100644 --- a/docker.env +++ b/docker.env @@ -1,12 +1,26 @@ -# Docker environment configuration -# Set this to use local Ollama instance running on host -# Note: Using Docker gateway IP instead of host.docker.internal for Linux compatibility -OLLAMA_HOST=http://172.18.0.1:11434 +# Docker environment configuration (passed via: docker compose --env-file docker.env ...) -# Alternative: Use containerized Ollama (uncomment and run with --profile with-ollama) +# Ollama running on the host. The compose files add +# extra_hosts: ["host.docker.internal:host-gateway"] so this also resolves on Linux. +OLLAMA_HOST=http://host.docker.internal:11434 + +# Alternative: containerized Ollama (start with: ./start-docker.sh container) # OLLAMA_HOST=http://ollama:11434 -# Other configuration +# Service wiring NODE_ENV=production +RAG_API_URL=http://rag-api:8001 + +# Browser-facing URLs. These are inlined into the frontend at build time. NEXT_PUBLIC_API_URL=http://localhost:8000 -RAG_API_URL=http://rag-api:8001 \ No newline at end of file +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 + +# Shared SQLite database, mounted into both backend and rag-api +DB_PATH=/app/backend/chat_data.db +LANCEDB_PATH=/app/lancedb + +# Models +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B diff --git a/env.example.watsonx b/env.example.watsonx index 5c3f3f86..5a7c015f 100644 --- a/env.example.watsonx +++ b/env.example.watsonx @@ -1,11 +1,18 @@ # ==================================================================== -# LocalGPT Watson X Configuration Example +# localGPT Watson X Configuration Example # ==================================================================== -# This file shows how to configure LocalGPT to use IBM Watson X AI -# with Granite models instead of local Ollama. +# This file shows how to point localGPT's LLM calls at IBM watsonx.ai +# Granite models instead of a local Ollama server. # # Copy this file to .env and fill in your credentials: -# cp .env.example.watsonx .env +# cp env.example.watsonx .env +# +# (The filename has no leading dot on purpose: .gitignore ignores .env*, +# so a dotted example file could never be committed.) +# +# See WATSONX_README.md for what the backend switch does and does not +# cover. Short version: generation and all utility LLM calls move to +# Watson X; embeddings, reranking and sentence pruning stay local. # ==================================================================== # LLM Backend Selection @@ -15,17 +22,20 @@ LLM_BACKEND=watsonx # ==================================================================== # Watson X Credentials # ==================================================================== -# Get these from your IBM Cloud Watson X project: +# Get these from your IBM Cloud watsonx.ai project: # 1. Go to https://cloud.ibm.com/ -# 2. Navigate to Watson X AI service +# 2. Navigate to the watsonx.ai service # 3. Create or select a project -# 4. Get API key from IBM Cloud IAM -# 5. Copy project ID from project settings +# 4. Get an API key from IBM Cloud IAM +# 5. Copy the project ID from the project settings +# +# Both of the next two are mandatory - rag_system/factory.py raises +# "Watson X configuration incomplete" if either is empty. # Your IBM Cloud API key WATSONX_API_KEY=your_api_key_here -# Your Watson X project ID +# Your watsonx.ai project ID WATSONX_PROJECT_ID=your_project_id_here # Watson X service URL (default: us-south region) @@ -39,23 +49,41 @@ WATSONX_URL=https://us-south.ml.cloud.ibm.com # ==================================================================== # Model Configuration # ==================================================================== -# Granite models available on Watson X +# Use model ids that exist in your watsonx.ai instance and region. The +# values below are only the defaults hardcoded in rag_system/main.py. +# IBM's "Supported foundation models" documentation lists what is +# currently available. -# Main generation model for answering queries -# Options: -# - ibm/granite-13b-chat-v2 (recommended for chat) -# - ibm/granite-13b-instruct-v2 (for instructions) -# - ibm/granite-20b-multilingual (for multilingual) -# - ibm/granite-3b-code-instruct (for code) +# Answers and sub-answer composition. +# default: ibm/granite-13b-chat-v2 WATSONX_GENERATION_MODEL=ibm/granite-13b-chat-v2 -# Lightweight model for enrichment and routing -# Use a smaller model for better performance on simple tasks +# Routing, triage, query decomposition, contextual enrichment, document +# overviews and answer verification. These prompts expect JSON back, and +# the Watson X client cannot force JSON mode - pick a model that follows +# format instructions well. +# default: ibm/granite-8b-japanese WATSONX_ENRICHMENT_MODEL=ibm/granite-8b-japanese # ==================================================================== -# Optional: Ollama Configuration (fallback) +# Local models (used no matter which LLM_BACKEND is selected) +# ==================================================================== +# Embeddings run in-process from Hugging Face when the name contains a +# "/", and through Ollama otherwise. Changing this requires re-indexing. +# default: microsoft/harrier-oss-v1-0.6b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b + +# Reranker: the Qwen3 default loads through the in-repo QwenRerankerScorer, +# other models through the `rerankers` library; loaded lazily on the first +# reranked query (reranking is on by default). +# default: Qwen/Qwen3-Reranker-4B +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B + +# ==================================================================== +# Optional: Ollama Configuration # ==================================================================== -# These settings are used if LLM_BACKEND=ollama +# Still used when LLM_BACKEND=ollama, and always used by the backend +# gateway's direct-LLM fast path (backend/server.py) and by any Ollama +# embedding model. OLLAMA_HOST=http://localhost:11434 diff --git a/eval/BASELINE.md b/eval/BASELINE.md new file mode 100644 index 00000000..8aed79b3 --- /dev/null +++ b/eval/BASELINE.md @@ -0,0 +1,750 @@ +# Phase 0 baseline โ€” measured 2026-08-08 + +> **Erratum 2026-08-16.** As of arm G (2026-08-14) the shipped `default` +> profile enables reranking with `min_score: 0.5` / `min_keep: 3` / `top_k: 10` +> threshold selection, and arm H (2026-08-15) made pooled first-stage +> decomposition the default. Any repro command on this page that assumes a +> reranker-off stack must now pass `--no-rerank` explicitly. Also: until this +> date the harness's `build_config` replaced the profile's reranker block with +> a bare `top_k: None` one, so `final` metrics from runs before this change +> described a reorder-only stack that never shipped; the harness now mirrors +> the shipped selection. The measurements below are unchanged and remain the +> record of their runs. + +> **Historical.** This page describes the stack as it stood on 2026-08-08. +> The defaults it calls "shipped" (`Qwen/Qwen3-Embedding-4B`, +> `BAAI/bge-reranker-v2-m3`, reranking on) were replaced at the Phase 1 +> adoption gate on 2026-08-09 โ€” see [`DECISIONS.md`](DECISIONS.md). The +> measurements below are unchanged and still valid *for that configuration*; +> they are the baseline Phase 1 was measured against, not current behavior. + +Every number on this page came from a run executed on this machine on +2026-08-08 (UTC 2026-08-09). Nothing here is estimated, extrapolated or copied +from a leaderboard. Where something could not be measured, it says so. + +Raw outputs: `eval/results/baseline_rerank.json`, `eval/results/baseline_norerank.json`, +`eval/results/judge_v{1,2,3}_*.json` (all git-ignored โ€” re-run the commands below +to regenerate them). + +## Configuration under test + +| | | +|---|---| +| Embedder | **`Qwen/Qwen3-Embedding-0.6B`** (1024-dim). The **shipped default is `Qwen/Qwen3-Embedding-4B`** (2560-dim); the 0.6B was used here because the 4B weights are not in the local HF cache (checked: `~/.cache/huggingface/hub/` holds `models--Qwen--Qwen3-Embedding-0.6B` and no 4B) and Phase 0 explicitly does not download new models. **These numbers are therefore not the shipped default's numbers.** | +| Reranker | `BAAI/bge-reranker-v2-m3`, loaded through the `rerankers` library as a cross-encoder โ€” the shipped default | +| Generation model | `qwen3.5:9b` (smoke test only) | +| Utility model | `qwen3.5:4b` (gold-query generation, groundedness judge, verifier in the smoke test) | +| Profile | `PIPELINE_CONFIGS["default"]`, with **contextual enrichment OFF, document overviews OFF, late chunking OFF, context expansion OFF, decomposition OFF, verification OFF, synthesis skipped** (rationale in `README.md`) | +| Chunking | docling chunker, `chunk_size = 512` tokens (what the HTTP path sends) | +| Retrieval | `hybrid` (LanceDB FTS + vector, RRF-fused), `k = 20` first-stage candidates | +| Seeds | `random`, `numpy`, `torch` all seeded with `20260808`; corpora and queries iterated in sorted order | +| Machine | Apple M2 Max, 96 GB, macOS 15.5, torch 2.4.1 on **MPS**, Python 3.12.12 | +| Libraries | transformers 4.51.0, rerankers 0.10.0, lancedb 0.36.0, docling 2.118.1, Ollama 0.32.6 | + +## Corpora and gold set + +| Corpus | Source | Files | Chunks indexed | Gold queries | +|--------|--------|-------|----------------|--------------| +| `atlas7` | `eval/corpora/atlas7_service_manual.pdf` (planted facts, fictional) | 1 | **1** | 24 | +| `hr` | `eval/corpora/northwind_leave_policy.pdf` (synthetic, fictional) | 1 | **2** | 24 | +| `docs` | `Documentation/*.md` minus `improvement_plan.md` / `research_roadmap.md` | 12 | **313** | 24 | +| `mixed` | all of the above in one table | 14 | **316** | 72 | + +**Gold set: 72 rows, all 72 verified, 0 discarded.** 72 (query, anchor) pairs were +generated from hand-authored dimension tuples, then checked one by one: + +| Verification verdict | Count | +|---|---| +| accepted verbatim | 39 | +| rescued (model emitted `{"question": โ€ฆ}` or a truncated payload; question hand-written from the anchor) | 6 | +| rewritten (wrong premise, too vague to be answerable, or the query restated the expected string) | 27 | +| discarded as unanswerable | **0** | + +Both automated gates passed with no exclusions: + +* Gate 1 โ€” `eval/corpora/verify_facts.py`: **76/76** planted-fact strings present in their source document. +* Gate 2 โ€” reachability after conversion + chunking: **24/24, 24/24, 24/24, 72/72** rows reachable; `coverage_failures` is empty in both results files. + +## Retrieval โ€” first-stage recall and nDCG@10 + +The `--no-rerank` run and the reranked run produce **identical first-stage +numbers on all 144 query evaluations** (checked per query, not just in +aggregate): same seeds, same sorted iteration, same ranking twice. That is the +determinism check. + +**First stage (recall is the headline metric here; nDCG@10 is over the +first-stage ordering):** + +| Corpus | n | chunks | recall@5 | recall@10 | recall@20 | nDCG@10 | +|--------|---|--------|----------|-----------|-----------|---------| +| `atlas7` | 24 | 1 | 1.000 | 1.000 | 1.000 | 1.000 | +| `hr` | 24 | 2 | 1.000 | 1.000 | 1.000 | 0.954 | +| `docs` | 24 | 313 | 0.750 | 0.917 | 0.958 | 0.515 | +| **`mixed`** | **72** | **316** | **0.917** | **0.972** | **0.986** | **0.805** | + +**After the `bge-reranker-v2-m3` cross-encoder** (same 20 candidates, reordered): + +| Corpus | recall@5 | recall@10 | recall@20 | nDCG@10 | ฮ” nDCG@10 | +|--------|----------|-----------|-----------|---------|-----------| +| `atlas7` | 1.000 | 1.000 | 1.000 | 1.000 | 0.000 | +| `hr` | 1.000 | 1.000 | 1.000 | 1.000 | +0.046 | +| `docs` | 0.750 | 0.958 | 0.958 | 0.731 | **+0.216** | +| **`mixed`** | **0.917** | **0.986** | **0.986** | **0.903** | **+0.098** | + +**Read `mixed` as the baseline.** `atlas7` and `hr` are 1 and 2 chunks, so `k=20` +sweeps the entire document and their recall is 1.0 by construction โ€” those two +rows test the plumbing, not the retriever. + +The reranker is where the measurable quality is: **+0.216 nDCG@10 on the only +corpus with real distractors**, and +0.098 on the mixed table. It also moves +recall, because recall@5 and recall@10 are prefixes of a list it reordered: +16 of the 144 evaluations change their recall vector after reranking (docs +recall@10 0.917 โ†’ 0.958, mixed 0.972 โ†’ 0.986). recall@20 is unchanged by +construction โ€” reranking cannot add candidates, only reorder the 20 it was +given. + +### By dimension (mixed + per-corpus runs pooled, n = 144 query evaluations) + +| Slice | n | recall@10 | nDCG@10 after rerank | +|-------|---|-----------|----------------------| +| difficulty = easy | 88 | 0.955 | 0.916 | +| difficulty = hard | 56 | 1.000 | 0.891 | +| type = factoid | 72 | 0.972 | 0.936 | +| type = comparative | 18 | 0.889 | 0.955 | +| type = negative | 28 | 1.000 | 0.886 | +| type = procedural | 26 | 1.000 | 0.815 | + +`hard` scoring above `easy` on recall is not a paradox to celebrate โ€” with 144 +evaluations the difference is 4 queries. Treat these slices as diagnostics, not +findings. + +### Latency (per query, wall clock, MPS) + +| Corpus | first stage mean | first stage p90 | rerank mean | rerank p90 | +|--------|-----------------|-----------------|-------------|------------| +| `atlas7` (1 chunk) | 158 ms | 210 ms | 176 ms | 107 ms | +| `hr` (2 chunks) | 125 ms | 132 ms | 293 ms | 216 ms | +| `docs` (313 chunks) | 142 ms | 150 ms | 1788 ms | 1783 ms | +| `mixed` (316 chunks) | 147 ms | 173 ms | 1741 ms | 1817 ms | + +The `atlas7` and `hr` rerank means exceed their p90 because the very first +rerank call in the process pays the cross-encoder's `from_pretrained()` load +(~2.5 s) and drags the mean up. **The cross-encoder is ~12x the cost of the +whole first stage** on a 20-candidate list โ€” that is the number Phase 1.1's +reranker A/B has to beat or justify. + +### Where retrieval actually fails today + +On the `docs` corpus, 8 of 24 gold queries scored below 0.5 nDCG@10 after +reranking and 1 more missed entirely. The same 9 reappear in `mixed`, joined by +one `atlas7` query the reranker demotes. + +| Query | Symptom | +|-------|---------| +| `docs_d08` "What does the enricher do when the model returns an almost-empty summary?" | first-stage recall@5 = 0, @10 = 0, @20 = 1 โ€” the answer-bearing chunk sits at rank 11โ€“20. The reranker rescues it to rank 1โ€“5, which is the single clearest case for keeping a cross-encoder | +| `docs_d16` "Does this project build an ANN indexโ€ฆ" | recall = 0 at every k. It is a `match: "all"` comparative and only one of its two anchors is ever retrieved; nDCG@10 is 1.000 because the anchor it did find was ranked first | +| `docs_d07`, `d09`, `d13`, `d14`, `d15`, `d19`, `d21` | the answer-bearing chunk is retrieved but ranked 4thโ€“10th, so nDCG@10 lands at 0.32โ€“0.39 | + +Three queries are **worse** after reranking than before it: `docs_d15` +(1.000 โ†’ 0.316), `docs_d09` (0.431 โ†’ 0.333), `docs_d21` (0.500 โ†’ 0.387), plus +`atlas7_a16` in the mixed table (1.000 โ†’ 0.431). On `docs` recall@5 the +cross-encoder promotes four queries into the top 5 (`d08`, `d10`, `d19`, `d22`) +and demotes four out of it (`d09`, `d14`, `d15`, `d17`) โ€” a net zero at @5, and a +clear win at @10. It is a net win in aggregate and a loss on some paraphrased +queries: exactly the situation Phase 1.1's reranker A/B exists to arbitrate. + +## Groundedness judge + +`qwen3.5:4b`, `format="json"`, `think: false`, binary verdict. 20 hand-built +cases in `eval/judge_validation.jsonl`: 10 grounded, 10 subtly ungrounded (wrong +number, transposed part code, swapped entities, unsupported addition). + +Three prompts were run. **v1 passed the gate on its first run**, so no iteration +was forced; v2 and v3 were then run as an ablation. + +| Prompt | TP | FN | TN | FP | unparseable | TPR | TNR | agreement | +|--------|----|----|----|----|-------------|-----|-----|-----------| +| **v1 (default)** | **10** | **0** | **10** | **0** | **0** | **1.00** | **1.00** | **1.00** | +| v2 (stricter, claim-by-claim) | 5 | 5 | 10 | 0 | 0 | 0.50 | 1.00 | 0.75 | +| v3 (stricter + explicit procedure) | 10 | 0 | 10 | 0 | 0 | 1.00 | 1.00 | 1.00 | + +**Gate (โ‰ฅ90% agreement): PASSED by v1 and v3. v2 fails at 75%.** v1 ships because +it ties v3 and is the shorter prompt. v1 was run twice (as `--prompt-version v1` +and again as the default after being promoted) and produced the identical +confusion matrix both times โ€” the judge is nondeterministic in principle, but it +did not wobble on this set. + +The honest caveat: 20/20 on 20 cases has a 95% Wilson lower bound of **0.839** +overall and **0.723** for TPR and TNR individually. This says the judge is not +badly broken; it does not say it is 100% accurate. v2's result is the useful +finding โ€” telling a 4B model to be *more* rigorous made it reject five correct +answers (it faulted `g08_sabbatical` for quoting the approver's title, and +`g05_warranty_length` for paraphrasing "above 8 percent" as "more than 8 +percent"). Before this judge gates anything in Phase 2, the validation set +should grow past 20. + +## End-to-end smoke + +`eval/smoke_e2e.py`, both services started as child processes against a temp +SQLite DB and a temp LanceDB directory, driven over HTTP, torn down afterwards. + +**Result: 25/25 assertions passed, exit code 0, 243.7 s wall clock.** + +| # | Assertion | Result | +|---|-----------|--------| +| 1 | both services became healthy (`:8001/health`, `:8000/health`) | PASS | +| 2 | `POST /indexes//upload` accepted the PDF | PASS | +| 3 | `POST /indexes//build` returned 200 with no `error` key | PASS | +| 4 | `POST /sessions//indexes/` linked them | PASS | +| 5โ€“8 | **q1** "What pressure does the brew boiler operate at during extraction?" โ†’ answer contains `9.2` ยท `source_documents` non-empty (1) ยท `[Confidence: 100%]` present ยท `message_count == 2` | PASS ร—4 | +| 9โ€“12 | **q2** "Which sensor part should be replaced when error code E11 appears?" โ†’ `TS-71` ยท 1 source ยท `[Confidence: 100%]` ยท `message_count == 4` | PASS ร—4 | +| 13โ€“16 | **q3** "How long is the Atlas-7 parts warranty?" โ†’ `36` ยท 1 source ยท `[Confidence: 100%]` ยท `message_count == 6` | PASS ร—4 | +| 17โ€“20 | **q4** "Where is the serial number engraved?" โ†’ `drip tray` ยท 1 source ยท `[Confidence: 100%]` ยท `message_count == 8` | PASS ร—4 | +| 21 | `POST /sessions//messages/save` returned 200 | PASS | +| 22 | the saved assistant message reads back out of SQLite | PASS | +| 23 | `source_documents` round-trip in `metadata.source_documents` | PASS | +| 24 | `steps` round-trip in `metadata.steps`, in order | PASS | +| 25 | `message_count == 10` after the saved turn | PASS | + +Teardown removed both child processes (SIGTERM, exit `-15`), the uploaded file +the gateway wrote into `shared_uploads/`, and the temp directory. + +All four verifier confidence tags came back at 100%. Do not read that as +calibration โ€” the roadmap (2.4) already flags `[Confidence: N%]` as UX, not a +measurement, and 4 identical maxima on 4 easy questions is exactly what an +uncalibrated self-report looks like. + +## Reproducing every number above + +```bash +cd /path/to/localGPT + +# gold-set gate 1 +.venv/bin/python eval/corpora/verify_facts.py + +# retrieval, with the cross-encoder (this also runs gate 2) +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ + .venv/bin/python eval/run_eval.py --corpus all \ + --json-out eval/results/baseline_rerank.json + +# retrieval, first stage only +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ + .venv/bin/python eval/run_eval.py --corpus all --no-rerank \ + --json-out eval/results/baseline_norerank.json + +# judge +.venv/bin/python eval/judge.py --validate --prompt-version v1 +.venv/bin/python eval/judge.py --validate --prompt-version v2 +.venv/bin/python eval/judge.py --validate --prompt-version v3 + +# end to end +.venv/bin/python eval/smoke_e2e.py +``` + +### Wall clock, as measured + +| Step | Time | +|------|------| +| Full retrieval eval, cold (rebuilds the `docs` and `mixed` indexes: 36.3 s + 40.9 s) | **281.0 s** | +| Full retrieval eval, warm (cached indexes, first stage only) | **19.3 s** | +| Judge validation, one prompt version, 20 cases | **21 s** | +| End-to-end smoke, including both service starts and 4 generations on `qwen3.5:9b` | **243.7 s** | + +Re-indexing is only triggered when a corpus file's size or mtime changes โ€” the +fingerprint lives in `eval/.eval_indexes///eval_.built.json`. + +## What this baseline does not tell you + +* Nothing about the **shipped 4B embedder**. Phase 1.2 must re-baseline before it + can claim a delta, because these indexes are 1024-dim. +* Nothing about **answer quality end to end** beyond the 4 smoke questions. The + retrieval metrics stop at the ranked chunk list; synthesis and verification are + deliberately excluded. +* Nothing about **contextual enrichment**, which is on in the shipped `default` + profile and off here. Turning it on changes the indexed text and would change + every number in the retrieval table. +* Nothing about **latency under load**. The RAG API is a single-threaded + `TCPServer`; everything above is single-user, single-request. + +--- + +# Phase 4 baseline (pre-implementation) โ€” measured 2026-08-09 + +The gate for [`Documentation/research_roadmap.md`](../Documentation/research_roadmap.md) +ยง Phase 4 (mechanisms adopted from *agentic-file-search*). **Nothing in Phase 4 +is implemented yet** โ€” this section measures the stack *as it ships today* on a +corpus and gold set built specifically to expose what 4.1/4.2/4.3 are supposed +to fix, so the post-implementation run has something to beat. + +Every number below came from a run executed on this machine on 2026-08-09. +Raw outputs (git-ignored, re-runnable โ€” commands at the end of this section): +`eval/results/phase4_baseline_acq.json`, +`eval/results/phase4_baseline_acq_docs.json`, +`eval/results/phase4_baseline_acq_docs_k{5,3}.json`, +`eval/results/phase4_regression_mixed.json`. + +## Configuration under test โ€” the current shipped defaults + +| | | +|---|---| +| Embedder | `microsoft/harrier-oss-v1-0.6b` (1024-dim, the shipped default since the Phase 1 adoption gate โ€” [`DECISIONS.md`](DECISIONS.md)) | +| Reranker | **off**, matching the shipped `default` profile | +| Evidence-sufficiency retry (2.1) | **on** (`--retry profile`, `min_top_score 0.12`) | +| Profile | `PIPELINE_CONFIGS["default"]` with enrichment / overviews / late chunking / context expansion / decomposition / verification off, as everywhere else in this harness | +| Chunking | docling chunker, `chunk_size = 512` | +| Retrieval | `hybrid` (LanceDB FTS + vector, RRF-fused), `k = 20` unless a row says otherwise | +| Machine | Apple M2 Max, macOS (Darwin arm64), torch 2.4.1 on **MPS**, Python 3.12.12, lancedb 0.36.0, docling 2.118.1, transformers 4.51.0 | + +## The new corpus: `acq` + +`eval/corpora/acquisition/` โ€” 10 interlinked synthetic M&A documents (TechCorp +acquires StartupXYZ), reused verbatim from the user's own +[PromtEngineer/agentic-file-search](https://github.com/PromtEngineer/agentic-file-search) +at `data/test_acquisition/`. They matter because they are the only corpus here +whose documents **reference each other**: 54 catalogued pointers of the form +`Document: `, `Exhibit A - Financial Terms`, `Schedule 1 - IP Assets`. +That link graph is what roadmap 4.2 (cross-reference hop) and 4.3 (overview +prefilter) act on, and neither planted-fact PDF nor `Documentation/*.md` has one. + +| Corpus | Files | Chunks | Gold queries | +|--------|-------|--------|--------------| +| `acq` | 10 PDFs (20 pages) | **13** | 24 | +| `acq+docs` | those 10 + `Documentation/*.md` minus the two excluded files | **373** | 48 (24 `acq` + 24 `docs`) | + +`eval/corpora/acquisition.facts.json` catalogues **100 planted facts** and the +**54 cross-references** (52 resolving inside the corpus, 2 deliberately dangling +โ€” both files point at a "Document: Integration Plan" that does not exist). + +**Gold set: 24 rows in `eval/goldset/acquisition.jsonl`, all 24 verified, 0 discarded.** +Hand-authored, not model-generated โ€” 8 rows adapt questions from the source +repo's `TEST_QUESTIONS.md`, the other 16 are new. The rows carry the usual +`{topic, question_type, difficulty}` plus a new boolean **`requires_crossref`**: +true when the query's premise points at document A while the answer text lives +in document B, reachable from A only through an explicit reference. + +| Composition | Count | +|---|---| +| `requires_crossref = true` | **11** | +| `requires_crossref = false` (control) | 13 | +| multi-document (`expected` spans โ‰ฅ2 documents, `match: "all"`) | 4 | +| question type factoid / comparative / negative / procedural | 14 / 5 / 3 / 2 | +| difficulty easy / hard | 11 / 13 | + +Per-row verification tally, all mechanical (re-run with +`.venv/bin/python eval/verify_crossref_goldset.py`), over the 31 `expected` +strings in the 24 rows: + +| Check | Result | +|---|---| +| `expected` present in the document named in `expected_sources` | **31/31** | +| query does not contain its `expected` string verbatim (no leak) | **31/31** | +| `expected` occurs in **exactly one** document of the ten (so "answer lives in a different document" is a real claim) | **31/31** | +| `fact_ids` resolve to the sidecar, with matching text and source | **31/31** | +| `requires_crossref` / `multi_document` consistent with `anchor_doc` | **24/24** rows | +| Gate 1 (`verify_facts.py`), whole repo | **176/176** facts, **54/54** cross-reference cues | +| Gate 2 (reachability after conversion + chunking) | **24/24** on `acq`, **48/48** on `acq+docs`, `coverage_failures` empty | + +## Results โ€” first stage, shipped defaults + +| Corpus | slice | n | chunks | recall@5 | recall@10 | recall@20 | nDCG@10 (1st) | 1st ms mean | 1st ms p90 | +|---|---|---|---|---|---|---|---|---|---| +| `acq` | all | 24 | 13 | 0.958 | 1.000 | 1.000 | 0.810 | 292 | 207 | +| `acq` | **`requires_crossref=true`** | 11 | 13 | **1.000** | **1.000** | **1.000** | **0.748** | | | +| `acq` | control (`=false`) | 13 | 13 | 0.923 | 1.000 | 1.000 | 0.863 | | | +| `acq+docs` | all | 48 | 373 | 0.917 | 0.958 | 1.000 | 0.738 | 309 | 997 | +| `acq+docs` | **`requires_crossref=true`** | 11 | 373 | **1.000** | **1.000** | **1.000** | **0.748** | | | +| `acq+docs` | control (`=false`) | 13 | 373 | 0.923 | 0.923 | 1.000 | 0.796 | | | +| `acq+docs` | *(the 24 `acq` rows alone)* | 24 | 373 | 0.958 | 0.958 | 1.000 | 0.774 | | | +| `acq+docs` | *(the 24 `docs` rows alone)* | 24 | 373 | 0.875 | 0.958 | 1.000 | 0.703 | | | +| `acq+docs` | multi-document rows | 4 | 373 | 1.000 | 1.000 | 1.000 | 0.748 | | | + +By dimension, `acq+docs`, pooled (recall@10 / nDCG@10 first stage): + +| Slice | n | recall@10 | nDCG@10 (1st) | +|---|---|---|---| +| `requires_crossref = true` | 11 | 1.000 | 0.748 | +| `requires_crossref = false` | 13 | 0.923 | 0.796 | +| difficulty = easy | 25 | 1.000 | 0.820 | +| difficulty = hard | 23 | 0.913 | 0.650 | +| type = factoid | 25 | 0.960 | 0.773 | +| type = comparative | 8 | 1.000 | 0.706 | +| type = negative | 9 | 1.000 | 0.739 | +| type = procedural | 6 | 0.833 | 0.637 | + +The evidence-sufficiency retry fired on 1/24 `acq` queries (`acq_q12`, rewrite +not kept) and 8/48 `acq+docs` queries (5 rewrites kept). + +## The crossref slice is **not** weak โ€” and that is the finding + +The expectation going in was that `requires_crossref=true` would be the visibly +broken slice. **It is not.** On both corpora the crossref rows hit +recall@5/@10/@20 = 1.000 โ€” *better* than their own control (0.923). The only +gap is in ranking, and it is small: nDCG@10 0.748 vs 0.863 (`acq`) / 0.796 +(`acq+docs`), i.e. the answer-bearing chunk typically lands at rank 2โ€“3 instead +of rank 1. Seven of the eleven crossref rows score 0.500โ€“0.631 (`q13`, `q23`, +`q14`, `q17`, `q18`, `q19`, `q22`); the other four score 1.000. + +Tightening `k` does not reverse it either. Same corpus, same gold set, first +stage only: + +| `acq+docs`, k = | crossref recall@k | control recall@k | crossref nDCG@10 | control nDCG@10 | +|---|---|---|---|---| +| 20 | 1.000 | 0.923 | 0.748 | 0.796 | +| 5 | 0.909 | 0.923 | 0.723 | 0.841 | +| 3 | **1.000** | **0.692** | **0.791** | 0.741 | + +Three honest reasons this corpus does not reproduce the "cross-references are +invisible to embeddings" failure at the first stage, all of them properties of +the measurement rather than of the retriever: + +1. **The deal room is 13 chunks.** With `k = 20` the first stage sweeps every + chunk of every document, exactly the saturation caveat that already applies + to `atlas7` and `hr`. `acq+docs` adds 360 distractor chunks, but they are + localGPT documentation โ€” topically disjoint from an M&A deal room, so they + compete weakly. +2. **The references are echoed at both ends.** "Exhibit A - Financial Terms" + appears in the Acquisition Agreement *and* in the Financial Adjustments Memo; + "Document: Risk Assessment Memo" appears in seven files. The hybrid FTS leg + therefore resolves most pointers lexically, without needing a hop. +3. **A crossref query still names its subject.** These queries state document + A's premise *and* ask about B's subject matter, which is what an honest + user question looks like โ€” but it hands the retriever lexical signal from + both ends of the reference. + +What this means for Phase 4, stated as a limitation rather than a conclusion: +**first-stage recall on `acq` cannot by itself decide item 4.2.** 11 rows on a +13-chunk corpus is a small measurement, and the slice is already at ceiling. +When 4.2 lands, the comparison worth making is (a) nDCG@10 on the crossref +slice, which has 0.25 of headroom and is the number in the table above, and +(b) an end-to-end measurement of *which document gets cited*, which this +retrieval-only harness does not perform. A stricter reference-only gold set โ€” +queries that name the pointer and nothing about the target's content โ€” would be +the way to make the first-stage metric discriminative, and does not exist yet. + +## Where retrieval actually fails on this corpus today + +| Query | Symptom | +|---|---| +| `acq_q04` "What proportion of the target company's turnover comes from its single biggest client?" | recall = 0 at @5 and @10, 1 at @20 on `acq+docs`; nDCG@10 = 0.000. The only full miss. Fully paraphrased away from the document's vocabulary ("turnover"/"biggest client" vs "revenue"/"largest customer") โ€” a query-understanding failure, not a cross-reference one | +| `acq_q12` "How large is the target's workforceโ€ฆ" | nDCG@10 = 0.316 (`acq`) / 0.431 (`acq+docs`); the one `acq` query the evidence-sufficiency retry fires on, and the rewrite was not kept | +| `acq_q01` "What is the total purchase priceโ€ฆ" | nDCG@10 = 0.500 โ€” the Financial Adjustments Memo's restated price outranks the Agreement's own definition | +| `acq_q13`, `q23`, `q14`, `q17`, `q18`, `q19`, `q22` | crossref rows whose answer chunk is retrieved but ranked 2ndโ€“3rd (nDCG@10 0.500โ€“0.631) | + +## Regression check: the pre-existing corpora are untouched + +Adding `acq` changed no shared code path โ€” `corpus_files()` gained list-valued +globs (existing corpora pass a string), corpus keys are slugged for the +filesystem (no existing key contains `+`), and `by_dimension` skips rows without +a `requires_crossref` key, which is all of them outside `acq`. `mixed` +re-measured on this tree: + +| | n | chunks | recall@5 | recall@10 | recall@20 | nDCG@10 (1st) | +|---|---|---|---|---|---|---| +| `mixed`, this run | 72 | **363** | 0.958 | 0.986 | 1.000 | **0.898** | +| `mixed`, `DECISIONS.md` ยง4 | 72 | 317 | 0.944 | 0.958 | 1.000 | 0.913 | + +The two are not comparable and the difference is not a regression: `mixed` +contains live `Documentation/*.md`, which grew from 317 to 363 chunks between +the two runs. Only compare `mixed` runs made against the same tree. + +## Reproducing every number in this section + +```bash +cd /path/to/localGPT + +# gate 1 โ€” planted facts and cross-reference cues really are in the PDFs +.venv/bin/python eval/corpora/verify_facts.py + +# row-level gate for the hand-authored gold set (the 31/31 tallies above) +.venv/bin/python eval/verify_crossref_goldset.py + +# gate 2 only +.venv/bin/python eval/run_eval.py --corpus acq --coverage-only + +# the two headline runs (shipped defaults: harrier, reranker off, retry on) +.venv/bin/python eval/run_eval.py --corpus acq \ + --json-out eval/results/phase4_baseline_acq.json +.venv/bin/python eval/run_eval.py --corpus acq+docs \ + --json-out eval/results/phase4_baseline_acq_docs.json + +# the k sweep behind the crossref-vs-k table +.venv/bin/python eval/run_eval.py --corpus acq+docs --k 5 \ + --json-out eval/results/phase4_baseline_acq_docs_k5.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --k 3 \ + --json-out eval/results/phase4_baseline_acq_docs_k3.json + +# regression check on the tracked corpus +.venv/bin/python eval/run_eval.py --corpus mixed \ + --json-out eval/results/phase4_regression_mixed.json +``` + +| Step | Wall clock, as measured | +|---|---| +| `acq` index build, cold (10 PDFs โ†’ 13 chunks) | 9.9 s | +| `acq` eval, warm | 8.8 s | +| `acq+docs` eval, cold (373-chunk index build included) | 59.9 s | +| `mixed` eval, warm | 43.0 s | + +## What this Phase 4 baseline does not tell you + +* **Nothing about 4.1, 4.4, 4.5 or 4.6.** Document escalation is a synthesis-time + behaviour, the filter DSL needs a `filters` argument that does not exist yet, + token accounting is an SSE field, and `ask` mode is a CLI entry point. None of + them is a first-stage retrieval metric, and none is measured here. +* **Nothing about 4.3 with overviews on.** This harness disables document + overviews (one LLM call per document, and it changes indexed text). The + overview-prefilter A/B will need that switched back on, which makes it the one + Phase 4 item that cannot reuse these exact indexes. +* **Nothing about answer quality.** Same boundary as the rest of this page: the + metrics stop at the ranked chunk list. +* **24 queries, one machine, one corpus of 13 chunks.** One query is 0.042 of + any recall figure on `acq`. Treat every slice here as a diagnostic. + +--- + +# Final-candidate-list metrics (added 2026-08-09) + +Everything above this line scores the **first stage** โ€” +`retrieve_candidates()["first_stage"]`. That is the right number for the +retriever, and it is the wrong number for roadmap item 4.2: the cross-reference +hop *appends* to `retrieve_candidates()["documents"]` and never mutates +`first_stage`, so a first-stage-only harness reports a flat line for the hop no +matter how well it works. + +`eval/run_eval.py` now scores **both** lists on every query. + +| metric family | source | meaning | +|---|---|---| +| `recall@k`, `ndcg@10_first_stage` | `out["first_stage"]` | the retriever's ordering. **Unchanged** โ€” every number above this line still means what it meant. | +| `recall@k_final`, `ndcg@10_final` | `out["documents"]` | post-rerank **and** post-cross-reference-hop: the list the answer stage would see | +| `ndcg@10_reranked`, `recall_reranked` | `out["documents"]` minus the `via_crossref` rows | still post-rerank / **pre**-hop, so it stays comparable with every earlier decision file | + +The results table prints the final family to the right of a `|` bar, plus a +`hop q` column (how many queries hopped). New CLI toggles, same shape as +`--retry`: `--crossref-hop {profile,on,off}` and +`--overview-prefilter {profile,off,boost,restrict}`; `profile` means whatever +`main.py` says, which is OFF for both today. + +## The invariant + +With reranking off and the hop off, `documents` **is** `first_stage`. Every run +checks this per query (chunk-id sequence, plus both metric families) and records +the verdict in the results JSON as `final_vs_first_stage_invariant`. + +``` +.venv/bin/python eval/run_eval.py --corpus mixed --retry off \ + --crossref-hop off --overview-prefilter off \ + --json-out eval/results/phase4_finalmetric_regression_mixed.json +``` + +``` +corpus n chunks R@5 R@10 R@20 nDCG@10 nDCG@10 | R@5 R@10 R@20 nDCG@10 hop q 1st ms + (1st) (rerank) | (fin) (fin) (fin) (final) +------------------------------------------------------------------------------------------------------------------------- +mixed 72 363 0.944 0.972 1.000 0.887 n/a | 0.944 0.972 1.000 0.887 0 124 + +invariant โœ… final == first_stage on all 72 queries (rerank OFF, crossref hop OFF) โ€” chunk-id order and both metrics +``` + +First stage identical to the tracked retry-off baseline (0.944 / 0.972 / 1.000, +nDCG@10 first-stage 0.887); final equals first-stage on all 72 queries. + +## First 4.2 A/B โ€” measured, and negative + +> **Superseded 2026-08-09 by the rebuilt-index runs at the bottom of this page.** +> Every arm in this subsection ran against indexes built *before* +> `rag_system/indexing/crossref.py` learned numeric-prefix-stripped filename +> aliases, so **none of the acquisition corpus's references resolved** and the hop +> could not fire on it. The subsection is kept because it is the measurement that +> found the resolver bug; it is not current evidence about the hop. + +All arms `--retry off`, reranker off, cached indexes (`acq` 13 chunks, +`acq+docs` 373 chunks), so the two arms of each pair differ only in the flag. + +| corpus | k | arm | slice | n | nDCG@10 (1st) | **nDCG@10 (final)** | queries that hopped | +|---|---|---|---|---|---|---|---| +| `acq` | 20 | hop off | all | 24 | 0.8101 | **0.8101** | 0 | +| `acq` | 20 | hop **on** | all | 24 | 0.8101 | **0.8101** | 0 | +| `acq` | 20 | hop off | `requires_crossref=true` | 11 | 0.7477 | **0.7477** | 0 | +| `acq` | 20 | hop **on** | `requires_crossref=true` | 11 | 0.7477 | **0.7477** | 0 | +| `acq+docs` | 20 | hop off | all | 48 | 0.7194 | **0.7194** | 0 | +| `acq+docs` | 20 | hop **on** | all | 48 | 0.7194 | **0.7194** | **7** | +| `acq+docs` | 20 | hop off | `requires_crossref=true` | 11 | 0.7477 | **0.7477** | 0 | +| `acq+docs` | 20 | hop **on** | `requires_crossref=true` | 11 | 0.7477 | **0.7477** | 0 | + +Recall is identical across every pair as well (`acq` 0.958 / 1.000 / 1.000; +`acq+docs` 0.854 / 0.896 / 0.958), first-stage and final alike. `--k 5` and +`--k 3` were also run on `acq`, both arms, and still fired zero hops. + +Why: **none of the acquisition corpus's 34 extracted cross-references resolve to +a target document** (`exhibit a`, `schedule 1`, `section 4.1` โ€” the Exhibits and +Schedules are sections *inside* `01_acquisition_agreement.pdf`, and resolution is +filename-based). The 7 hops on `acq+docs` all come from `Documentation/*.md` +title matches, and `hit_expected_source = 0`, `hopped_chunk_relevant = 0` โ€” the +hop pulled nothing gold. Full analysis and the raw index dump: +[`decisions/phase4-eval-final-metric.md`](decisions/phase4-eval-final-metric.md). + +## Reproducing this subsection + +```bash +cd /path/to/localGPT + +# regression + invariant on the tracked corpus +.venv/bin/python eval/run_eval.py --corpus mixed --retry off \ + --crossref-hop off --overview-prefilter off \ + --json-out eval/results/phase4_finalmetric_regression_mixed.json + +# the 4.2 A/B (each pair differs only in --crossref-hop) +.venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop off \ + --overview-prefilter off --json-out eval/results/phase4_42_acq_hop_off.json +.venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop on \ + --overview-prefilter off --json-out eval/results/phase4_42_acq_hop_on.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop off \ + --overview-prefilter off --json-out eval/results/phase4_42_acqdocs_hop_off.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop on \ + --overview-prefilter off --json-out eval/results/phase4_42_acqdocs_hop_on.json + +# the k sweep that shows low k does not rescue the hop on `acq` +for k in 5 3; do for arm in off on; do + .venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop $arm \ + --overview-prefilter off --k $k \ + --json-out eval/results/phase4_42_acq_k${k}_hop_${arm}.json +done; done +``` + +**Determinism protocol** for any comparison run: pass `--retry off` (the retry is +an LLM reformulation and is nondeterministic), never pass an empty-string env var +(`EMBEDDING_MODEL=` breaks the run), and do not edit anything under +`Documentation/` between arms โ€” the docs corpus is live in `mixed`, `docs` and +`acq+docs`, and a doc diff moves the numbers. + +--- + +# Rebuilt-index baselines โ€” measured 2026-08-09 (supersede the two rows above) + +## Why the rebuild + +Cross-references are stamped into chunk metadata **at index time**. The +`acq` and `acq_plus_docs` eval indexes were built before +`rag_system/indexing/crossref.py` existed, and certainly before the gate's +**resolver fix** โ€” each known document is now additionally registered under its +numeric-prefix-stripped name, so `08_regulatory_approval.pdf` also answers to +`"regulatory approval"`, which is how the acquisition PDFs actually refer to each +other (`phase4-crossref-prefilter.md` ยง *Gate correction (2026-08-09)*). Both +index directories were deleted and rebuilt on 2026-08-09. + +## What the rebuild produced + +Read out of the built LanceDB tables (`metadata` โ†’ `["metadata"]["crossrefs"]`), +not from the build log: + +| index | chunks | chunks with crossrefs | refs | **resolved** | documents linked | self-edges | +|---|---|---|---|---|---|---| +| `acq` | 13 | 11 | 68 | **34** | **9 of 10** | none | +| `acq+docs` | 373 | 70 | 215 | **93** | 21 | none | + +Previously: **0 resolved on `acq`**. The 34 that still do not resolve are the +Exhibits and Schedules โ€” sections *inside* `01_acquisition_agreement.pdf`, not +separate files โ€” plus bare `section N` forms. A filename-based resolver cannot +reach them and correctly leaves them `target_doc: null`. + +## Re-baseline โ€” zero drift + +`--retry off`, reranker off, hop off, prefilter off, k = 20. Gold coverage +24/24 and 48/48, `coverage_failures` empty, chunk counts unchanged (13 / 373), +`final == first_stage` invariant โœ… on all 72 queries across the two runs. + +| corpus | slice | n | chunks | recall@5 | recall@10 | recall@20 | nDCG@10 (1st) | vs. previous retry-off figure | +|---|---|---|---|---|---|---|---|---| +| `acq` | all | 24 | 13 | 0.958 | 1.000 | 1.000 | **0.8101** | identical | +| `acq` | `requires_crossref=true` | 11 | 13 | 1.000 | 1.000 | 1.000 | **0.7477** | identical | +| `acq` | control (`=false`) | 13 | 13 | 0.923 | 1.000 | 1.000 | **0.8628** | identical | +| `acq+docs` | all | 48 | 373 | 0.854 | 0.896 | 0.958 | **0.7194** | identical | +| `acq+docs` | `requires_crossref=true` | 11 | 373 | 1.000 | 1.000 | 1.000 | **0.7477** | identical | +| `acq+docs` | control (`=false`) | 13 | 373 | 0.769 | 0.769 | 0.846 | **0.7731** | identical | + +Identical to four decimals on every cell, which is the expected result and the +point of running it: extraction writes chunk *metadata* only โ€” `text` and +`vector` are untouched โ€” so adding 34 resolved references cannot move a +first-stage ranking. These are the numbers any future 4.2/4.3 arm must be +compared against. + +## Headline of the re-measured A/B (full matrix in the decision file) + +| | | +|---|---| +| Hop fires now | 0 โ†’ **24/24** queries on `acq` at k=3 and k=5; **33/48** on `acq+docs` | +| At the shipped **k = 20** | **inert**: 0 hops on `acq` (13-chunk corpus < candidate budget), 14 hops on `acq+docs` with every metric bit-identical | +| On the `requires_crossref` slice | **0/11 hops hit a gold source document at any k on either corpus; no metric moved** | +| Where it gains | only the `requires_crossref=false` control slice at k=3/k=5 (e.g. `acq` k=3 recall@10 final 0.692 โ†’ 0.846) | +| Budget-matched vs. simply raising `k` | wins **1 of 4** cells | +| Harm | none โ€” the hop only appends, recall never fell | + +## 4.3 is now measurable โ€” `--overviews on` + +The overview prefilter reads an `.npz` sidecar that only an overview-enabled +build writes, and this harness had overviews hard-off. `eval/run_eval.py` gained +**`--overviews {off,on}`** (default `off`, unchanged behaviour): `on` enables +document overviews + the embedded sidecar and redirects `overview_path` into the +corpus's own index directory (`<corpus>_ov`), so the sidecar is owned by the +index and the repo's shared `index_store/overviews/` is never written. Verified +chunk-for-chunk identical to a normal build (373/373 chunk ids, text identical, +max abs vector delta **0.0**). + +| corpus | arm | recall@5 | recall@10 | recall@20 | nDCG@10 (1st) | +|---|---|---|---|---|---| +| `acq+docs` | off | 0.854 | 0.896 | 0.958 | **0.7194** | +| `acq+docs` | boost | 0.812 | 0.917 | 0.958 | **0.7017** | +| `acq+docs` | restrict | 0.812 | 0.854 | **0.896** | **0.6951** | +| `mixed` | off | 0.944 | 0.972 | 1.000 | **0.8873** | +| `mixed` | boost | 0.889 | 0.972 | 1.000 | **0.8662** | +| `mixed` | restrict | 0.889 | 0.931 | **0.944** | **0.8740** | + +`boost` is a large gain on the heterogeneous slice (`acq` control rows of +`acq+docs`: nDCG@10 0.7731 โ†’ 0.8790) and a loss on `mixed`, where twelve of +fifteen documents are localGPT documentation and the overviews carry no +discriminating signal. **`restrict` loses four queries their answer document +entirely on each corpus** (recall@20 1 โ†’ 0) โ€” the harm check the previous +decision file asked for. + +Caveat specific to 4.3: overview text is LLM-generated. All three arms of each +comparison read the same sidecar, so each comparison is exact, but a rebuild of +`<corpus>_ov` will produce different overviews and can move these numbers with no +code change. + +Full matrix, hop-precision columns, budget-matched controls, per-query harm +traces and the proposed adopt/reject/hold calls: +[`decisions/phase4-retrieval-benchmarks.md`](decisions/phase4-retrieval-benchmarks.md). + +## Reproducing this section + +```bash +cd /path/to/localGPT + +rm -rf eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq \ + eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq_plus_docs + +# rebuild + re-baseline (these are the two hop-off k=20 arms) +.venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop off \ + --overview-prefilter off \ + --json-out eval/results/phase4_w3_rebuild_acq_hop_off.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop off \ + --overview-prefilter off \ + --json-out eval/results/phase4_w3_rebuild_acqdocs_hop_off.json + +# 4.3 arms (the first run per corpus builds the _ov index: 23 / 15 LLM calls) +for m in off boost restrict; do + .venv/bin/python eval/run_eval.py --corpus acq+docs --overviews on --retry off \ + --crossref-hop off --overview-prefilter $m \ + --json-out eval/results/phase4_w3_43_acqdocs_ov_${m}.json + .venv/bin/python eval/run_eval.py --corpus mixed --overviews on --retry off \ + --crossref-hop off --overview-prefilter $m \ + --json-out eval/results/phase4_w3_43_mixed_ov_${m}.json +done +``` + +The full 4.2 k-sweep and the budget-matched controls are in +[`decisions/phase4-retrieval-benchmarks.md`](decisions/phase4-retrieval-benchmarks.md) ยง 7. + +**Latency**: not reported for any run in this section. A concurrent agent shared +the Ollama instance throughout, so every wall-clock figure is contended. diff --git a/eval/DECISIONS.md b/eval/DECISIONS.md new file mode 100644 index 00000000..c3ba8b2a --- /dev/null +++ b/eval/DECISIONS.md @@ -0,0 +1,202 @@ +> **Erratum 2026-08-16.** This page's "reranking OFF by default" decision no +> longer describes the shipped `default` profile: arm G (2026-08-14) turned +> reranking ON with `min_score: 0.5` / `min_keep: 3` / `top_k: 10` threshold +> selection, and arm H (2026-08-15) made pooled first-stage decomposition the +> default (`compose_from_sub_answers: False, pooled_first_stage: True`). Repro +> commands that assume a reranker-off stack must now pass `--no-rerank` +> explicitly. Note also that the eval harness's `build_config` replaced the +> whole reranker block with `top_k: None` until this date, so `final` metrics +> from runs before this change described a reorder-without-selection stack that +> never shipped; the harness now keeps the profile's selection and overrides +> only the model name. The measurements below are unchanged โ€” they are records +> of their runs. + +# Phase 1 component decisions โ€” adopted 2026-08-09 + +The decision gate the roadmap asks for +([`Documentation/research_roadmap.md`](../Documentation/research_roadmap.md) ยง1): +*"adopt each only on a measured win; record adopted/rejected + numbers in +`eval/DECISIONS.md`."* This is that record. + +Three A/Bs fed it, each with its own evidence page: + +| Roadmap item | Investigation | Outcome | +|---|---|---| +| 1.1 reranker | [`decisions/reranker.md`](decisions/reranker.md) | **Reranking OFF by default**; `Qwen/Qwen3-Reranker-4B` is what the toggle loads | +| 1.2 embedder | [`decisions/embedder.md`](decisions/embedder.md) | **`microsoft/harrier-oss-v1-0.6b` adopted as the default embedder**, query-side instruction prefix on | +| 1.3 GLM-OCR | [`decisions/glm-ocr-spike.md`](decisions/glm-ocr-spike.md) | **GO-LATER โ€” no code change**; three defects and a missing eval corpus block adoption | + +1.1 and 1.2 could not be decided independently: the reranker's value depends +entirely on how good the first stage is. They were therefore re-measured +*jointly* and decided in one re-index window, together with the two index-format +hazards the embedder audit surfaced. + +--- + +## 1. The joint matrix that decided it + +`mixed` corpus (72 queries, 316 chunks โ€” the only corpus with real distractors), +`docs` corpus (24 queries). First stage is hybrid RRF at k=20; every reranker +reorders the same 20 candidates. Latency is per query on an M2 Max (MPS) with a +shared GPU, so treat it as an order of magnitude, not a benchmark. + +| Stack | mixed nDCG@10 | docs nDCG@10 | added latency | +|---|---|---|---| +| **harrier-0.6b, first stage only** โ† **shipped** | **0.915** | **0.759** | โ€” (~140 ms total) | +| harrier-0.6b + `BAAI/bge-reranker-v2-m3` | 0.892 | 0.701 | **+1.6 s โ€” net negative** | +| harrier-0.6b + `Qwen/Qwen3-Reranker-4B` | 0.977 | 0.932 | +12.7 s | +| *(the previously-measured stack: `BASELINE.md`)* Qwen3-Embedding-0.6B + bge | 0.908 | 0.747 | +1.6 s | + +Three things follow, and all three are counter-intuitive enough to be worth +stating plainly: + +1. **harrier's first stage alone beats the whole previously-measured stack** + (0.915 vs 0.908 on `mixed`) at a fraction of the latency. +2. **The cheap cross-encoder now *hurts*.** โˆ’0.022 nDCG@10 on `mixed` and โˆ’0.058 + on `docs`, for ~1.6 s per query. bge-reranker-v2-m3's famous +0.232 on `docs` + was largely a repair job on a weak first stage; improve the first stage and + the repair becomes damage. +3. **The good reranker is still a real win โ€” and still too slow to default on.** + +0.062 nDCG@10 on `mixed` and +0.173 on `docs`, for ~12.7 s per query and + 7.5 GB of resident weights, on a single-user server. That is worth paying on + demand, not on every message. + +One caveat on the table itself: every cell in it was measured **before** the +cosine-normalization fix in ยง2 (item 7), on unnormalized vectors. Normalization moves +the shipped first-stage cell by โˆ’0.004 nDCG@10 โ€” smaller than the gaps the +decision rests on, but it means the shipped number is 0.911/0.913, not 0.915. +ยง2 and ยง4 carry the shipped measurement. + +Supporting detail lives in the two decision pages: the embedder comparison +(harrier 0.915 vs Qwen3-Embedding-4B 0.875 vs Qwen3-Embedding-0.6B 0.805 on +`mixed`, first stage) in `decisions/embedder.md` ยง2 and its gate section, and the +three-way reranker A/B (bge 0.9077, Qwen3-Reranker-0.6B 0.9289, Qwen3-Reranker-4B +0.9825 on the older 0.6B-embedder first stage) in `decisions/reranker.md` ยง3. + +--- + +## 2. What shipped + +| # | Change | Where | +|---|---|---| +| 1 | Default embedder โ†’ `microsoft/harrier-oss-v1-0.6b` (MIT, 1024-dim, 1.2 GB) | `rag_system/main.py::EXTERNAL_MODELS` | +| 2 | Query-side instruction prefix stays on for harrier and the Qwen3-Embedding family | `rag_system/indexing/representations.py`, `rag_system/pipelines/retrieval_pipeline.py::_query_instruction` | +| 3 | `default` profile ships `reranker.enabled = False` | `rag_system/main.py::PIPELINE_CONFIGS` | +| 4 | Default reranker model โ†’ `Qwen/Qwen3-Reranker-4B`, loaded lazily only when the toggle is on | `rag_system/main.py::EXTERNAL_MODELS`, `rag_system/pipelines/retrieval_pipeline.py::_get_ai_reranker` | +| 5 | UI "AI reranker" toggle now defaults off, matching the profile | `src/components/ui/session-chat.tsx` | +| 6 | Per-table embedder identity marker + guard (index time and query time) | `rag_system/indexing/embedders.py`, `rag_system/retrieval/retrievers.py` | +| 7 | Vectors L2-normalized at write and query time, gated on that marker | same two files | +| 8 | Default table name `text_pages_v3` โ†’ `text_pages_v4` | `rag_system/main.py` | +| 9 | GLM-OCR | *nothing* โ€” GO-LATER, by design | + +### Why the identity marker (6) + +The pre-existing guard compared **vector width only**. `harrier-oss-v1-0.6b` and +`Qwen3-Embedding-0.6B` are **both 1024-dim**, so swapping between exactly those +two โ€” the swap this adoption makes people likely to perform โ€” passed the guard +and appended mutually unintelligible vectors to a live table, silently. Each +table now records the embedding model that wrote it plus a `normalized` flag, in +the table's Arrow schema metadata (lancedb 0.36.0 round-trips it; a sidecar +`<db_path>/table_meta/<table>.json` is written if a future version does not). +Indexing into, or querying, a table whose recorded model differs from the +configured one now raises with a rebuild instruction instead of returning +nonsense. This is the pipeline-level guard and it covers **every** table, +including CLI-built ones; the backend's per-named-index `embedding_model` +metadata is unchanged and complementary. + +### Why normalization (7) โ€” and what it did *not* buy + +Both model cards specify cosine similarity, but vectors were stored unnormalized +and searched with LanceDB's default L2 metric, so the ranking was neither cosine +nor intended. L2 ordering equals cosine ordering exactly when every vector is +unit length, so vectors are now L2-normalized on both sides. + +Measured on `mixed`, this is **a wash, not a win** โ€” it was adopted for +conformance with the model cards, not for a number. Both arms below were run +back to back against the same 316-chunk corpus snapshot, with normalization +neutered at *both* write and query time for the control: + +| mixed, harrier-0.6b, first stage | recall@5 | recall@10 | recall@20 | nDCG@10 | +|---|---|---|---|---| +| unnormalized (control โ€” reproduces `decisions/embedder.md` exactly) | 0.917 | 0.958 | 0.986 | 0.915 | +| **L2-normalized (shipped)** | 0.931 | 0.944 | 0.986 | **0.911** | + +(The control is not a repo flag: it was produced by monkey-patching +`l2_normalize` to the identity in both `rag_system/indexing/embedders.py` and +`rag_system/retrieval/retrievers.py` and pointing the harness at a throwaway +index directory. There is no supported way to turn normalization off, and there +should not be.) + +One query gained at recall@5, one lost at recall@10, nDCG@10 moved โˆ’0.004. +On 72 queries that is inside the noise floor `decisions/embedder.md` ยง8 sets for +itself. **The 0.915 headline in `decisions/embedder.md` was measured on +unnormalized vectors and does not transfer unchanged to the shipped stack** โ€” +ยง4 below has the number that does. + +Normalization is a property of the *table*, not of the config: new tables are +normalized, and a table without the marker is queried the old way with a warning +recommending a rebuild, so no index ever mixes the two conventions. The default +table name moved to `text_pages_v4` so the shipped default starts clean. + +--- + +## 3. What stays opt-in (and how to switch it on) + +| Option | When it is the right choice | Switch | +|---|---|---| +| `Qwen/Qwen3-Embedding-4B` | multilingual or long-context (32K) corpora โ€” capabilities this English, digital-born gold set does not exercise. Keep the query prefix on: it measured **+0.059 nDCG@10** for this model. | `EMBEDDING_MODEL=Qwen/Qwen3-Embedding-4B` + rebuild the index | +| `Qwen/Qwen3-Reranker-4B` | quality-first sessions where ~12.7 s per query is acceptable | UI "AI reranker" toggle, or `reranker.enabled: true` | +| `BAAI/bge-reranker-v2-m3` | the low-latency legacy option; only pays off with a **weaker** embedder than the current default โ€” on top of harrier it is net negative | `RERANKER_MODEL=BAAI/bge-reranker-v2-m3` + enable reranking | +| GLM-OCR for scanned PDFs | not yet โ€” see `decisions/glm-ocr-spike.md` | no code exists; nothing to switch | + +Rejected outright: **`Qwen/Qwen3-Reranker-0.6B`** โ€” +0.021 nDCG@10 over bge +(one to two queries out of 72) bought with 1.5โ€“2.8ร— the latency, while *losing* +recall@10 on both corpora. `decisions/reranker.md` ยง7 has the full argument. + +--- + +## 4. Post-adoption verification + +Re-run any time; the reranker follows the shipped profile, so a bare run +measures the shipped stack: + +```bash +.venv/bin/python eval/run_eval.py --corpus mixed \ + --json-out eval/results/post_adoption_mixed.json +``` + +Latest result, on this tree after the documentation updates that shipped with +this decision โ€” `mixed`, 317 chunks, 72 queries, first stage only, embedder +reported as `microsoft/harrier-oss-v1-0.6b` and reranker as `(disabled)`: + +| | recall@5 | recall@10 | recall@20 | nDCG@10 (1st stage) | mean ms | p90 ms | +|---|---|---|---|---|---|---| +| `mixed` | **0.944** | **0.958** | **1.000** | **0.913** | 120.4 | 200.1 | + +Gold coverage: 72/72 rows reachable, zero coverage failures. +Raw output: `eval/results/post_adoption_mixed.json`. + +Note that the `docs` and `mixed` corpora are live `Documentation/*.md` content, +so editing the documentation moves these numbers โ€” the run above is 317 chunks +because this decision's own doc updates grew the corpus by one chunk, which is +why it differs slightly from the 316-chunk numbers in ยง2. Only compare runs made +against the same tree. + +--- + +## 5. Limits of this evidence + +Carried forward verbatim in spirit from the two decision pages, because they +apply to the adopted defaults just as much as to the experiments: + +* **72 English queries on one machine, over digital-born corpora.** A 0.014 + recall delta is one query. Only the large gaps (+0.110 embedder, +0.173 + reranker on `docs`) are outside the noise. +* **Latency was measured on a shared GPU** and is indicative, not a benchmark. +* **Nothing here measures answer quality** โ€” these are retrieval metrics. + Groundedness is a separate harness (`eval/judge.py`). +* **`docs_d09` and `docs_d17`** degrade under every reranker tested. They are a + query-understanding problem and no default here fixes them. +* **Qwen3-Reranker latency was never tuned** (batch size 8, 2048-token cap, + 20 candidates). The 12.7 s figure is untuned, and tuning it is unmeasured + work โ€” not a promise. diff --git a/eval/README.md b/eval/README.md new file mode 100644 index 00000000..ba7f1745 --- /dev/null +++ b/eval/README.md @@ -0,0 +1,374 @@ +# `eval/` โ€” Phase 0 evaluation harness + +The gate for [`Documentation/research_roadmap.md`](../Documentation/research_roadmap.md). +Nothing in Phase 1 or Phase 2 lands without a measured delta against these numbers. +The numbers themselves live in [`BASELINE.md`](BASELINE.md). + +Everything here is read-only with respect to the product: no file under +`rag_system/`, `backend/` or `src/` is modified or imported-and-monkeypatched. +The harness calls the shipped pipeline objects directly. + +--- + +## Layout + +| Path | What it is | +|------|------------| +| `corpora/atlas7_service_manual.pdf` | 2-page planted-fact PDF (fictional espresso machine). 21 facts. | +| `corpora/northwind_leave_policy.pdf` | 3-page synthetic HR handbook, generated by `make_hr_handbook.py`. 24 facts. | +| `corpora/*.facts.json` | Sidecars: `{id, topic, expected, summary}` per planted fact. `expected` is the verbatim answer-bearing substring. | +| `corpora/repo_docs.facts.json` | 31 prose anchors into `Documentation/*.md` โ€” the real, heterogeneous, third corpus. Referenced in place, never copied. | +| `corpora/acquisition/*.pdf` | 10 interlinked synthetic M&A documents (2 pages each) โ€” the **cross-reference** corpus, added for roadmap Phase 4. Reused verbatim from [PromtEngineer/agentic-file-search](https://github.com/PromtEngineer/agentic-file-search) `data/test_acquisition/`. | +| `corpora/acquisition.facts.json` | 100 planted facts **plus** a `cross_references` block: the 54 `Document:` / `Exhibit` / `Schedule` pointers between those PDFs, `to: null` for the 2 deliberately dangling ones. | +| `corpora/rfc/*.txt` | 23 interlinked IETF RFCs (QUIC / HTTP-3 family), byte-for-byte from rfc-editor.org โ€” the only corpus whose naming and referencing conventions this project did not author. `corpora/rfc/download.py` re-fetches them and checks the link graph; `corpora/rfc/MANIFEST.md` is the rationale. Sidecar: `corpora/rfc/rfc.facts.json` (26 facts). | +| `corpora/verify_facts.py` | Gate 1: asserts every `expected` string exists in its source document, and every cross-reference cue in its `from` document. Recurses into subdirectories, so the `rfc` sidecar is covered too. | +| `corpora/make_hr_handbook.py` | Regenerates the HR PDF and re-checks its sidecar. | +| `build_goldset.py` | Reverse-generates one query per dimension tuple with `qwen3.5:4b`. One-shot. Covers the three Phase 0 corpora only โ€” `acquisition.jsonl` and `rfc.jsonl` are hand-authored. | +| `finalize_goldset.py` | Applies the recorded human verification pass; writes the committed gold set. | +| `verify_crossref_goldset.py` | Row-level gate for `goldset/acquisition.jsonl`: answer in its named document, no verbatim leak into the query, each answer string unique to one document, `requires_crossref` / `multi_document` consistent with `anchor_doc`, and `anchor_doc` itself a corpus document. | +| `verify_rfc_goldset.py` | The same row-level gate ported to `goldset/rfc.jsonl` (normalisation handles the RFCs' 72-column hard wrapping). | +| `goldset/<corpus>.jsonl` | **The gold set of record.** 120 rows: 24 each for `atlas7`, `hr`, `docs`, `acquisition` and `rfc`. Plus `multiturn.jsonl` (12 hand-authored multi-turn rows) โ€” no runner is wired for it yet. | +| `goldset/_generated/*.raw.jsonl` | Raw model output, kept for audit. | +| `run_eval.py` | Retrieval metrics: recall@5/10/20 first stage, nDCG@10 before and after rerank, the `*_final` family over the list the answer stage actually sees (post-rerank, post-crossref-hop), the `requires_crossref` slices, per-query latency, and the `final == first_stage` invariant check whenever nothing is allowed to reorder or append. | +| `judge.py` | Binary groundedness judge + its own validation harness. | +| `judge_validation.jsonl` | 20 hand-built cases, 10 grounded / 10 subtly ungrounded. | +| `judge_hard_cases.jsonl` | 18 real system answers, hand-adjudicated โ€” the judge screen that decides judge swaps. | +| `validate_judge_hard.py` | Scores a candidate judge against `judge_hard_cases.jsonl`; majority vote over an odd number of runs, parse errors reported as errors rather than coerced to votes. | +| `smoke_e2e.py` | Starts both services on temp storage, indexes the Atlas-7 PDF over HTTP, asserts answers/citations/persistence, tears down. | +| `.eval_indexes/` | Cached LanceDB indexes, keyed by embedder. Git-ignored, safe to delete. | +| `results/` | Run outputs (JSON + log). Git-ignored. | + +--- + +## Running it + +All commands are from the repo root, with the project venv. + +```bash +# gate 1 โ€” planted facts really are in the source documents +.venv/bin/python eval/corpora/verify_facts.py + +# gate 2 + retrieval metrics on the shipped defaults (reranking ON with +# threshold selection since arm G โ€” that is what ships, so that is what a bare +# run measures) +.venv/bin/python eval/run_eval.py --corpus all \ + --json-out eval/results/shipped_defaults.json + +# first-stage-only control arm +.venv/bin/python eval/run_eval.py --corpus all --no-rerank \ + --json-out eval/results/shipped_defaults_norerank.json + +# with a different reranker, for the reranker A/B โ€” naming one swaps the model +.venv/bin/python eval/run_eval.py --corpus all \ + --reranker Qwen/Qwen3-Reranker-4B \ + --json-out eval/results/rerank_ab.json + +# judge validation (TPR / TNR / confusion matrix) +.venv/bin/python eval/judge.py --validate + +# end-to-end, starts :8000 and :8001 as children on a temp DB, then kills them +.venv/bin/python eval/smoke_e2e.py +``` + +`run_eval.py` flags: `--embedder`, `--reranker`, +`--corpus {atlas7,hr,docs,mixed,acq,acq+docs,rfc,all}`, `--k`, `--chunk-size`, +`--no-rerank`, `--retry {profile,on,off}`, `--crossref-hop {profile,on,off}`, +`--overview-prefilter {profile,off,boost,restrict}`, `--overviews {off,on}`, +`--decompose`, `--aggregate {max,mean}`, +`--coverage-only`, `--force-reindex`, `--json-out`, `--verbose`. + +`acq` and `acq+docs` are the Phase 4 corpora (see below); `rfc` is the +real-document shakedown corpus (`decisions/rfc-shakedown-2026-08-13.md`). +`mixed` deliberately does **not** include any of them: it is the corpus every +Phase 0/1/2 number is quoted against and it has to keep meaning the same thing. + +Since roadmap Phase 2 the harness drives `RetrievalPipeline.retrieve_candidates()` +rather than calling `MultiVectorRetriever.retrieve()` itself, so first stage, +reranking **and the evidence-sufficiency retry** are the shipped code path. Two +consequences worth knowing: + +* `--retry` defaults to `profile`, i.e. **on**, matching what ships. It is the one + stage in the list below that the harness does *not* disable, because it is + conditional: it only fires on queries whose first pass found weak evidence. Use + `--retry off` for the control arm of an A/B. When it fires it makes one + enrichment-model call, so a bare run is no longer fully deterministic โ€” the + *firing set* is deterministic, the rewrites are not. +* `first_stage_ms` now covers first stage + rerank + any retry on that query, and + `rerank_ms` is reported as 0 on this path. Per-query timings are no longer + split by stage. + +`--decompose` runs `QueryDecomposer` once per gold row (cached under +`eval/.eval_indexes/_subqueries/`, keyed by corpus + decomposer model + prompt +version, so the on and off arms differ only in whether the sub-queries are +*used*) and hands them to the **rerank** stage. The first stage always uses the +full original query โ€” that is the whole of roadmap item 2.2. With the rerank +stage off there is no consumer for sub-queries, so `--decompose` is skipped +entirely โ€” literally a no-op, not even the LLM calls. + +`--embedder` defaults to whatever `EMBEDDING_MODEL` resolves to in +`rag_system/main.py`, i.e. the shipped default `microsoft/harrier-oss-v1-0.6b` +unless the env var overrides it. Each embedder gets its own index directory +(`eval/.eval_indexes/<slug>/`) and the embedder is part of the cache +fingerprint, so two embedders can neither share nor inherit an index. + +The rerank stage follows the shipped profile, which has it **on since arm G +(2026-08-14)**: `Qwen/Qwen3-Reranker-4B` with `min_score: 0.5` / `min_keep: 3` / +`top_k: 10` threshold selection ([`DECISIONS.md`](DECISIONS.md) records the +earlier off-by-default call it supersedes). When the stage is on the harness +keeps that selection block and overrides only the model name, so the `final` +metrics describe the list the answer stage actually sees โ€” not the +reorder-without-selection stack a bare block would measure. `--reranker +<model>` swaps the model; `--no-rerank` forces the stage off. + +Regenerating the gold set (only when the corpora or the dimension table change): + +```bash +.venv/bin/python eval/build_goldset.py --corpus all # LLM writes the questions +.venv/bin/python eval/finalize_goldset.py # applies the recorded verification pass +``` + +--- + +## How the gold set is built + +Reverse-generated the structured way, not "give me some questions": + +1. **Dimension tuples** are hand-authored in `build_goldset.py`: (anchor fact(s)) + ร— (question type: `factoid` / `procedural` / `comparative` / `negative`) + ร— (difficulty: `easy` / `hard`). 24 tuples per corpus. + * `negative` means a question about a restriction, exclusion, limit or + invalidating condition **that the document does state** โ€” not a question the + document cannot answer. Every gold row is answerable. + * `easy` reuses the document's vocabulary; `hard` paraphrases away from it, so + the lexical leg cannot carry the query. +2. **`qwen3.5:4b` phrases the question** for each tuple, through the repo's own + `OllamaClient`. The label โ€” the answer-bearing substring โ€” is fixed by the + tuple, never by the model. +3. **Every pair is verified by hand** (see `finalize_goldset.py`, which records + the verdict and the reason for each row, and `goldset/*.jsonl`, which carries + them per row under `verification`). +4. **Gate 1** (`verify_facts.py`): the expected string is in the source document. +5. **Gate 2** (`run_eval.py`, run automatically before scoring): the expected + string survives conversion + chunking into at least one indexed chunk. Rows + that fail are printed and land in `coverage_failures` in the results JSON โ€” + they are never silently dropped. + +Gold relevance is **answer-bearing text**, not chunk ids, so the set survives +re-chunking, a different chunk size, and an embedder swap. That is the whole +reason it can gate Phase 1's A/B tests. + +### The `acquisition` gold set is different + +`goldset/acquisition.jsonl` is **hand-authored**, not model-generated, because +its whole purpose is a property no dimension tuple can express: *the answer is +in a different document from the one the question points at.* Eight of its 24 +rows adapt questions from the source repo's `TEST_QUESTIONS.md`; the rest are +new. Its rows carry two extra fields on top of the shared schema: + +| Field | Meaning | +|---|---| +| `dimensions.requires_crossref` | **bool.** `true` when the query's premise names or paraphrases document A while at least one `expected` string lives in document B, reachable from A only through an explicit reference. This slice is roadmap item 4.2's metric. | +| `expected_sources`, `anchor_doc`, `multi_document` | The document holding each `expected` string, the document the query's premise points at (`null` when the query names none), and whether the row's answer spans more than one document. `requires_crossref` is exactly `anchor_doc is not None and any(source != anchor_doc)`. | + +Every row was verified mechanically by `verify_crossref_goldset.py`, and the +counts are recorded in [`BASELINE.md`](BASELINE.md) ยง *Phase 4 baseline*: the +`expected` string is in its named document, the query does not contain it +verbatim, and the string occurs in **exactly one** of the ten documents โ€” +without that last check "the answer is in another document" would not be a +claim you could measure. Re-run it any time: + +```bash +.venv/bin/python eval/verify_crossref_goldset.py +``` + +`by_dimension` therefore gains `requires_crossref=true` / `=false` buckets, and +`summary.<corpus>.crossref` / `.crossref_control` carry the same slice per +corpus (pooling across corpora would double-count rows that appear in both +`acq` and `acq+docs`). Rows without the key โ€” every row outside `acquisition` โ€” +are skipped, so no pre-existing slice moves. + +## Metric definitions + +* `recall@k` โ€” over the first-stage (pre-rerank) ranking. `match: "any"` rows hit + when one of the top-k chunks contains an expected string; `match: "all"` rows + (the comparatives) hit only when the top-k *union* covers every expected + string. Mean over queries. +* `nDCG@10` โ€” binary per-chunk relevance (chunk contains any expected string), + DCG over the top 10 divided by the IDCG of the same candidate set sorted + ideally. A query whose candidates contain nothing relevant scores 0, so a + first-stage miss is never hidden by the ranking metric. +* Consequence worth knowing: a `match: "all"` query that retrieved only one of + its two anchors scores `recall = 0` but can still score `nDCG@10 = 1.0` โ€” + ranking was perfect, coverage was not. Read the two columns together. +* Latency is wall-clock per query, split into first stage and rerank, on the + machine in `BASELINE.md`. It is not a benchmark of anything but this laptop. + +## What the harness turns off, and why + +`run_eval.py` starts from `PIPELINE_CONFIGS["default"]` and disables: + +| Stage | Why | +|-------|-----| +| Contextual enrichment | One LLM round-trip per chunk. Nondeterministic, and it changes the indexed text, which would make the substring labels ambiguous. | +| Document overviews | One LLM round-trip per document; only feeds triage, which the harness does not exercise. | +| Late chunking | Doubles the vectors and merges sibling text into hits, which would smear the substring labels across chunk boundaries. | +| Context expansion | Same smearing problem; and when the reranker runs it is a no-op anyway (see `retrieval_pipeline.md`). | +| Query decomposition, verification, synthesis | Downstream of the two metrics being measured, and each adds LLM calls. Decomposition can be switched back on for the item-2.2 A/B with `--decompose`. | + +The evidence-sufficiency retry is deliberately **not** in that list โ€” see the +`--retry` note above. + +`chunk_size` is 512, matching what the HTTP path sends (`api_server.py`), not the +1500-token CLI default. + +## Caveats that the numbers do not show on their own + +* **`atlas7`, `hr` and `acq` in isolation are saturated.** They are 1, 2 and 13 + chunks, so `k=20` sweeps the entire corpus and recall@k is 1.0 by construction. + Their isolated rows are a smoke check on the plumbing, not a retrieval + measurement. `mixed` is the row to track for Phase 0โ€“2, `acq+docs` for Phase 4. +* **`acq+docs` is a weak stress test, honestly labelled.** Its 360 distractor + chunks are localGPT documentation โ€” topically disjoint from an M&A deal room, + so they compete far less than same-domain distractors would. Read the crossref + slice with the caveats in `BASELINE.md` ยง *The crossref slice is not weak*. +* **The `docs` corpus is live repo content.** Editing `Documentation/*.md` + changes the corpus, invalidates the cached index, and moves the baseline. + `improvement_plan.md` and `research_roadmap.md` are excluded for exactly this + reason (`DOCS_EXCLUDE` in `run_eval.py`) โ€” they are the files this harness's + own bookkeeping edits. +* **Anchors are reused across dimensions.** The two PDFs have ~20 facts each and + carry 24 queries each, so a few facts back two differently-typed questions. +* **The judge is an LLM.** Its TPR/TNR are measured on 20 hand-built cases, which + is a small sample; treat the interval, not the point estimate, as the truth. + +--- + +## Benchmarking roadmap Phase 4 + +Phase 4 (`Documentation/research_roadmap.md` ยง *Ideas adopted from +agentic-file-search*) is **implemented behind profile flags that ship OFF** โ€” +`retrieval.crossref_hop`, `retrieval.overview_prefilter` and +`retrieval.document_escalation` are all `enabled: False` in the `default` +profile. The harness drives them with `--crossref-hop`, `--overview-prefilter` +and `--overviews`; this section is the protocol the A/Bs follow, written before +the code so the comparison cannot be retro-fitted to a result. The +pre-implementation numbers are recorded in [`BASELINE.md`](BASELINE.md) ยง +*Phase 4 baseline (pre-implementation)*; every "off" arm below should reproduce +them. + +Ground rules, the same three as every other gate in this harness: + +1. **Only `acq` and `acq+docs` move.** If a Phase 4 flag changes `mixed`, that + is a regression to explain, not a result to report. Re-run `--corpus mixed` + on the same tree for both arms โ€” the `docs` corpus is live repo content, so + comparing against a number measured on a different tree proves nothing. +2. **One flag at a time**, both arms in the same session, same embedder, same + `k`, same `chunk_size`. `--force-reindex` whenever the flag touches index + content or chunk metadata (4.2 and 4.3 do). +3. **Report the `requires_crossref=true` slice next to its control**, never + alone: the whole claim of item 4.2 is a *gap* between the two, and both move + when the retriever changes. + +### 4.2 โ€” `retrieval.crossref_hop` + +The headline Phase 4 A/B, and the one the `acquisition` corpus exists for. + +The switch is `--crossref-hop {profile,on,off}`: `apply_phase4_settings()` +writes `retrieval.crossref_hop.enabled` into the config exactly the way +`apply_retry_setting()` writes the retry block โ€” in `run_eval.py`, not as an +env var, and not by editing the shipped profile. Then: + +```bash +# off (must reproduce BASELINE.md ยง Phase 4) +.venv/bin/python eval/run_eval.py --corpus acq+docs --crossref-hop off --force-reindex \ + --json-out eval/results/p4_crossref_off.json +# on +.venv/bin/python eval/run_eval.py --corpus acq+docs --crossref-hop on --force-reindex \ + --json-out eval/results/p4_crossref_on.json +``` + +Read, in this order: **`summary["acq+docs"]["crossref"]["ndcg@10_first_stage"]` +against `["crossref_control"]`** (0.748 vs 0.796 today), then the crossref +recall vector, then `mixed` for collateral damage. Because the hop *adds* +candidates, also check that `candidates` per query has not silently grown past +`--k` โ€” a recall win bought by retrieving more chunks is not a ranking win. +The `k = 3` column in the baseline table is the sharpest available comparison: +at `k = 3` the crossref slice is at 1.000 recall today, so a hop that pays for +itself has to show up as nDCG, not recall. + +### 4.3 โ€” `retrieval.overview_prefilter` + +The only Phase 4 item that **cannot reuse the cached indexes**: the harness +disables document overviews unless `--overviews on` is given, and the prefilter +needs them. Both arms must therefore be run with overviews on, which costs one +LLM call per document at index time and makes the index build nondeterministic. +`--overviews on` builds into a separate `_ov` index directory and changes the +fingerprint, so it cannot clobber the tracked indexes. + +```bash +# both arms need overviews ON; --overview-prefilter picks the arm +.venv/bin/python eval/run_eval.py --corpus acq+docs --overviews on --force-reindex \ + --overview-prefilter off --json-out eval/results/p4_overview_off.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --overviews on --force-reindex \ + --overview-prefilter boost --json-out eval/results/p4_overview_on.json +``` + +Do not compare an overview-prefilter arm against the numbers in `BASELINE.md`: +those indexes have no overviews in them. The control arm must be re-measured. +`acq+docs` is the corpus that can show anything here โ€” `acq` alone is 10 +documents and 13 chunks, so "restrict to the top documents" has almost nothing +to restrict. + +### 4.1 โ€” `retrieval.document_escalation` + +**Not a first-stage retrieval metric.** Escalation reassembles a document for +*synthesis* after the 2.1 retry still lands weak, so `run_eval.py` cannot see +it: the harness stops at the ranked chunk list and never synthesises. Two +things it *can* contribute, and they are worth logging: + +* **The trigger set.** The retry already reports per query + (`summary.retry_fired` / `retry_fire_rate` / `retry_kept`; today 1/24 on + `acq`, 8/48 on `acq+docs`, and 32/48 at `--k 5`). Whatever condition 4.1 + escalates on is a subset of that, so `--retry on` vs `--retry off` bounds how + often escalation can fire before anyone writes it. +* **Everything else belongs in `smoke_e2e.py` and `judge.py`** โ€” answer contains + the fact, citation names the right document, groundedness verdict โ€” driven + over HTTP against a session indexed on `eval/corpora/acquisition/`. That is + the arm that measures 4.1, and it is answer quality, not recall. + +### 4.4 โ€” filter DSL (`filters` on /chat and /chat/stream) + +Measurable here only once `run_eval.py` can pass a `filters` argument through +`retrieve_candidates()`. The natural A/B on this corpus is a per-row +`filters` field in the gold set (e.g. `document = "07_nda.pdf"`) on the rows +whose answer document is unambiguous, then: + +```bash +.venv/bin/python eval/run_eval.py --corpus acq+docs \ + --json-out eval/results/p4_filters_off.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --filters \ + --json-out eval/results/p4_filters_on.json +``` + +The number that matters is nDCG@10 on the **control** slice, not the crossref +slice: a document filter derived from the query's anchor document is exactly +the wrong thing to apply to a crossref row, and a filter A/B that improves the +control while wrecking `requires_crossref=true` is the expected โ€” and +reportable โ€” outcome. + +### 4.5 โ€” per-query token/cost tracking + +Nothing to A/B: it adds fields to the SSE `complete` event, it does not change +retrieval. Assert the fields exist and are non-zero in `smoke_e2e.py`; do not +give it a row in a retrieval table. + +### 4.6 โ€” `ask <folder>` ephemeral mode + +Also not a `run_eval.py` job โ€” it is a CLI entry point that builds a throwaway +index. The check it needs is an equivalence one: `python -m rag_system.main ask +eval/corpora/acquisition "<query>"` should answer the `acq` gold rows the same +way the persistent index does. Add it to `smoke_e2e.py` as a subprocess +assertion over a handful of gold rows (start with the four `requires_crossref` +rows that already score 1.000, so a failure is unambiguous), and record the +wall clock โ€” the claim being tested is "an ephemeral index beats an ephemeral +agent", which is a latency claim as much as a quality one. diff --git a/eval/build_goldset.py b/eval/build_goldset.py new file mode 100644 index 00000000..3a485018 --- /dev/null +++ b/eval/build_goldset.py @@ -0,0 +1,237 @@ +"""Reverse-generate the gold query set from planted facts, one query per dimension tuple. + +Structured, not freeform: the dimension table below is hand-authored +(anchor fact(s) x question-type x difficulty), and the *only* thing the LLM does +is phrase a natural-language question for a tuple whose answer is already fixed. +That keeps the gold label (the answer-bearing substring) independent of the model +that wrote the question. + + .venv/bin/python eval/build_goldset.py --corpus all --out eval/goldset/_generated + +Writes ``<out>/<corpus>.raw.jsonl``. The *committed* gold set lives at +``eval/goldset/<corpus>.jsonl`` and is the human-verified edit of that raw file โ€” +see eval/README.md. This script is a one-shot generator, not part of the eval run. + +Gold relevance is defined by answer-bearing text, never by chunk id, so the gold +set survives re-chunking, a different chunk size, or an embedder swap. +""" + +import argparse +import json +import os +import re +import sys + +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))) + +from rag_system.utils.ollama_client import OllamaClient # noqa: E402 + +CORPORA_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "corpora") + +QUESTION_TYPES = ("factoid", "procedural", "comparative", "negative") +DIFFICULTIES = ("easy", "hard") + +TYPE_GUIDANCE = { + "factoid": "Ask for one specific value, name, part number or figure.", + "procedural": "Ask how to do something, or what steps to take. Phrase it the way an operator would.", + "comparative": "Ask a question that can only be answered by using BOTH source snippets together โ€” a difference, a contrast, or a combined total.", + "negative": ( + "Ask about a restriction, exclusion, limit, threshold or condition that " + "invalidates something. The answer must still be stated in the snippet โ€” " + "this is a negatively-framed question about the document, NOT a question " + "the document cannot answer." + ), +} + +DIFFICULTY_GUIDANCE = { + "easy": "Use the document's own vocabulary. A keyword search would plausibly find it.", + "hard": "Paraphrase. Avoid reusing the snippet's distinctive nouns and numbers, so that only a semantic match finds it.", +} + +PROMPT = """You write evaluation questions for a document-retrieval benchmark. + +SOURCE SNIPPET(S), verbatim from {doc_label}: +{snippets} + +Write ONE natural question that a real user of this document would ask, whose +answer is contained in the snippet(s) above. + +Question type: {qtype}. {type_guidance} +Difficulty: {difficulty}. {difficulty_guidance} + +Hard rules: +- The question must be answerable using ONLY the snippet(s) above. +- Do NOT include the answer in the question. +- One sentence. No preamble, no quotes around it. +- Do not mention "the snippet", "the document above" or "the text". + +Respond with JSON only: {{"query": "<the question>"}} +""" + +# --- Dimension table ------------------------------------------------------- +# (tuple_id, question_type, difficulty, [anchor fact ids], match) +# match "any": retrieving one anchor-bearing chunk counts as a hit. +# match "all": every anchor must appear in the retrieved set (comparatives). +DIMENSIONS = { + "atlas7": [ + ("a01", "factoid", "easy", ["atlas_brew_pressure"], "any"), + ("a02", "factoid", "easy", ["atlas_steam_pressure"], "any"), + ("a03", "factoid", "easy", ["atlas_brew_temperature"], "any"), + ("a04", "factoid", "easy", ["atlas_pump_rating"], "any"), + ("a05", "factoid", "easy", ["atlas_water_hardness"], "any"), + ("a06", "factoid", "easy", ["atlas_gasket_part"], "any"), + ("a07", "factoid", "easy", ["atlas_warranty_length"], "any"), + ("a08", "factoid", "easy", ["atlas_manufacturer"], "any"), + ("a09", "factoid", "easy", ["atlas_model_revision"], "any"), + ("a10", "factoid", "easy", ["atlas_serial_location"], "any"), + ("a11", "factoid", "hard", ["atlas_temperature_tolerance"], "any"), + ("a12", "factoid", "hard", ["atlas_e11_part"], "any"), + ("a13", "procedural", "easy", ["atlas_e42_prime", "atlas_e42_procedure"], "all"), + ("a14", "procedural", "easy", ["atlas_backflush"], "any"), + ("a15", "procedural", "hard", ["atlas_descale_interval", "atlas_water_hardness"], "all"), + ("a16", "procedural", "hard", ["atlas_e57"], "any"), + ("a17", "procedural", "hard", ["atlas_e23"], "any"), + ("a18", "comparative", "easy", ["atlas_brew_pressure", "atlas_steam_pressure"], "all"), + ("a19", "comparative", "hard", ["atlas_descale_interval", "atlas_gasket_interval"], "all"), + ("a20", "comparative", "hard", ["atlas_e11", "atlas_e23"], "all"), + ("a21", "negative", "easy", ["atlas_warranty_void"], "any"), + ("a22", "negative", "easy", ["atlas_water_hardness"], "any"), + ("a23", "negative", "hard", ["atlas_temperature_tolerance"], "any"), + ("a24", "negative", "hard", ["atlas_e23"], "any"), + ], + "hr": [ + ("h01", "factoid", "easy", ["hr_annual_below_g7"], "any"), + ("h02", "factoid", "easy", ["hr_annual_g7_plus"], "any"), + ("h03", "factoid", "easy", ["hr_sick_full_pay"], "any"), + ("h04", "factoid", "easy", ["hr_parental_length"], "any"), + ("h05", "factoid", "easy", ["hr_bereavement"], "any"), + ("h06", "factoid", "easy", ["hr_jury_duty"], "any"), + ("h07", "factoid", "easy", ["hr_public_holiday_count"], "any"), + ("h08", "factoid", "easy", ["hr_policy_id"], "any"), + ("h09", "factoid", "easy", ["hr_policy_owner"], "any"), + ("h10", "factoid", "easy", ["hr_sabbatical_length"], "any"), + ("h11", "factoid", "hard", ["hr_carryover_expiry"], "any"), + ("h12", "factoid", "hard", ["hr_sick_reduced_pay"], "any"), + ("h13", "factoid", "hard", ["hr_public_holiday_pay"], "any"), + ("h14", "procedural", "easy", ["hr_request_notice"], "any"), + ("h15", "procedural", "easy", ["hr_medical_certificate"], "any"), + ("h16", "procedural", "hard", ["hr_sabbatical_eligibility", "hr_sabbatical_notice", "hr_sabbatical_approver"], "all"), + ("h17", "procedural", "hard", ["hr_director_approval"], "any"), + ("h18", "comparative", "easy", ["hr_annual_below_g7", "hr_annual_g7_plus"], "all"), + ("h19", "comparative", "hard", ["hr_sick_full_pay", "hr_sick_reduced_pay"], "all"), + ("h20", "comparative", "hard", ["hr_carryover_cap", "hr_carryover_expiry"], "all"), + ("h21", "negative", "easy", ["hr_contractors_excluded"], "any"), + ("h22", "negative", "easy", ["hr_resignation_payout"], "any"), + ("h23", "negative", "hard", ["hr_parental_deadline"], "any"), + ("h24", "negative", "hard", ["hr_parental_blocks"], "any"), + ], + "docs": [ + ("d01", "factoid", "easy", ["docs_provence_model"], "any"), + ("d02", "factoid", "easy", ["docs_verifier_context_clamp"], "any"), + ("d03", "factoid", "easy", ["docs_triage_overview_cap"], "any"), + ("d04", "factoid", "easy", ["docs_embedding_dimensions"], "any"), + ("d05", "factoid", "easy", ["docs_overview_truncation"], "any"), + ("d06", "factoid", "easy", ["docs_direct_answer_length"], "any"), + ("d07", "factoid", "easy", ["docs_latechunk_cost"], "any"), + ("d08", "factoid", "easy", ["docs_enrichment_short_summary"], "any"), + ("d09", "factoid", "hard", ["docs_query_embed_cache"], "any"), + ("d10", "factoid", "hard", ["docs_graph_two_llm_calls"], "any"), + ("d11", "factoid", "hard", ["docs_verifier_zero_score"], "any"), + ("d12", "procedural", "easy", ["docs_reindex_required"], "any"), + ("d13", "procedural", "easy", ["docs_pruning_off_by_default"], "any"), + ("d14", "procedural", "hard", ["docs_chunk_size_layering"], "any"), + ("d15", "procedural", "hard", ["docs_txt_bypasses_docling"], "any"), + ("d16", "comparative", "easy", ["docs_no_ann_index", "docs_brute_force_vector"], "all"), + ("d17", "comparative", "hard", ["docs_verifier_cost", "docs_triage_utility_model"], "all"), + ("d18", "comparative", "hard", ["docs_no_overlap_knob", "docs_indexing_sequential"], "all"), + ("d19", "negative", "easy", ["docs_no_weighted_blend"], "any"), + ("d20", "negative", "easy", ["docs_no_citation_markers"], "any"), + ("d21", "negative", "easy", ["docs_triage_no_regex"], "any"), + ("d22", "negative", "hard", ["docs_triage_no_switch"], "any"), + ("d23", "negative", "hard", ["docs_expansion_filtered_out"], "any"), + ("d24", "negative", "hard", ["docs_dimension_mismatch_raises"], "any"), + ], +} + +SIDECARS = { + "atlas7": "atlas7_service_manual.facts.json", + "hr": "northwind_leave_policy.facts.json", + "docs": "repo_docs.facts.json", +} + +DOC_LABELS = { + "atlas7": "the Atlas-7 espresso machine service manual", + "hr": "the Northwind Robotics leave and absence policy handbook", + "docs": "the localGPT project's developer documentation", +} + + +def load_facts(corpus: str) -> dict: + with open(os.path.join(CORPORA_DIR, SIDECARS[corpus]), "r", encoding="utf-8") as fh: + return {f["id"]: f for f in json.load(fh)["facts"]} + + +def strip_think(text: str) -> str: + return re.sub(r"<think>.*?</think>", "", text, flags=re.S).strip() + + +def generate(corpus: str, client: OllamaClient, model: str) -> list: + facts = load_facts(corpus) + rows = [] + for tuple_id, qtype, difficulty, fact_ids, match in DIMENSIONS[corpus]: + anchors = [facts[fid] for fid in fact_ids] + snippets = "\n".join(f'{i + 1}. "{a["expected"]}" โ€” {a["summary"]}' for i, a in enumerate(anchors)) + prompt = PROMPT.format( + doc_label=DOC_LABELS[corpus], + snippets=snippets, + qtype=qtype, + type_guidance=TYPE_GUIDANCE[qtype], + difficulty=difficulty, + difficulty_guidance=DIFFICULTY_GUIDANCE[difficulty], + ) + resp = client.generate_completion(model=model, prompt=prompt, format="json") + raw = strip_think(resp.get("response", "") or "") + try: + query = json.loads(raw).get("query", "").strip() + except json.JSONDecodeError: + query = "" + rows.append({ + "id": f"{corpus}_{tuple_id}", + "corpus": corpus, + "query": query, + "expected": [a["expected"] for a in anchors], + "match": match, + "fact_ids": fact_ids, + "dimensions": {"topic": anchors[0]["topic"], "question_type": qtype, "difficulty": difficulty}, + "generator_model": model, + "raw_response": raw if not query else None, + }) + status = "ok " if query else "FAIL" + print(f"[{status}] {corpus}_{tuple_id} ({qtype}/{difficulty}): {query or raw[:120]!r}") + return rows + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--corpus", default="all", choices=["all", *sorted(DIMENSIONS)]) + parser.add_argument("--model", default=os.getenv("ENRICHMENT_MODEL", "qwen3.5:4b")) + parser.add_argument("--host", default=os.getenv("OLLAMA_HOST", "http://localhost:11434")) + parser.add_argument("--out", default=os.path.join(os.path.dirname(os.path.abspath(__file__)), "goldset", "_generated")) + args = parser.parse_args() + + os.makedirs(args.out, exist_ok=True) + client = OllamaClient(host=args.host) + corpora = sorted(DIMENSIONS) if args.corpus == "all" else [args.corpus] + + for corpus in corpora: + rows = generate(corpus, client, args.model) + path = os.path.join(args.out, f"{corpus}.raw.jsonl") + with open(path, "w", encoding="utf-8") as fh: + for row in rows: + fh.write(json.dumps(row, ensure_ascii=False) + "\n") + print(f"\nWrote {len(rows)} rows to {path}\n") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/corpora/acquisition.facts.json b/eval/corpora/acquisition.facts.json new file mode 100644 index 00000000..e0ae196e --- /dev/null +++ b/eval/corpora/acquisition.facts.json @@ -0,0 +1,168 @@ +{ + "corpus": "acq", + "documents_dir": "acquisition", + "generator": "reused verbatim from PromtEngineer/agentic-file-search @ data/test_acquisition/ (10 interlinked synthetic M&A PDFs + TEST_QUESTIONS.md), catalogued 2026-08-09", + "note": "Fictional companies, people and transaction. Ten documents that reference each other by name ('Document: <Title>'), by exhibit ('Exhibit A - Financial Terms') and by schedule ('Schedule 1 - IP Assets'). This is the multi-document, cross-referenced corpus roadmap Phase 4 items 4.1/4.2/4.3 need; the two planted-fact PDFs and the docs corpus have no inter-document references at all. Every 'expected' string is a verbatim substring of its named source PDF after whitespace normalisation (checked by eval/corpora/verify_facts.py), and every cross_references[].cue is a verbatim substring of its 'from' document.", + "known_inconsistencies": [ + "05_financial_adjustments.pdf says 'Total Cash Required at Closing: $44,630,000' (cash 28.33M + escrow 1.3M + stock 10M + earnout 5M is not internally consistent either) while 10_closing_checklist.pdf sums the same line items to 'Total at Closing: $39,630,000'. Both are catalogued as separate facts with their own source; no gold row asks for 'the' total, and the deliberate-contradiction question is out of scope for a retrieval metric.", + "Both 04_risk_assessment.pdf and 10_closing_checklist.pdf reference 'Document: Integration Plan', which does not exist in the corpus. Kept as-is: a dangling cross-reference is a realistic negative case for item 4.2 and is flagged with \"dangling\": true below." + ], + "facts": [ + {"id": "acq_agreement_date", "topic": "agreement_terms", "source": "01_acquisition_agreement.pdf", "expected": "entered into as of January 15, 2025", "summary": "The Acquisition Agreement is dated January 15, 2025."}, + {"id": "acq_purchase_price", "topic": "purchase_price", "source": "01_acquisition_agreement.pdf", "expected": "means $45,000,000 USD as detailed in Exhibit A", "summary": "The headline Purchase Price is $45,000,000, detailed in Exhibit A."}, + {"id": "acq_closing_date_defined", "topic": "timeline", "source": "01_acquisition_agreement.pdf", "expected": "means March 1, 2025, subject to conditions in Article IV", "summary": "The Closing Date is March 1, 2025, subject to Article IV conditions."}, + {"id": "acq_cash_at_closing", "topic": "purchase_price", "source": "01_acquisition_agreement.pdf", "expected": "$30,000,000 in cash at Closing", "summary": "Original structure: $30M cash at closing."}, + {"id": "acq_stock_component", "topic": "purchase_price", "source": "01_acquisition_agreement.pdf", "expected": "$10,000,000 in Buyer's common stock", "summary": "Original structure: $10M in buyer stock."}, + {"id": "acq_earnout_component", "topic": "purchase_price", "source": "01_acquisition_agreement.pdf", "expected": "$5,000,000 in earnout payments", "summary": "Original structure: $5M earnout."}, + {"id": "acq_employee_matters_ref", "topic": "cross_reference", "source": "01_acquisition_agreement.pdf", "expected": "governed by Schedule 3 - Employee Transition Plan", "summary": "Employee matters are pushed to Schedule 3."}, + {"id": "acq_nda_execution_date", "topic": "confidentiality", "source": "01_acquisition_agreement.pdf", "expected": "executed between the parties on October 1, 2024", "summary": "Article V points at an NDA executed 1 Oct 2024."}, + {"id": "acq_buyer_signatory", "topic": "parties", "source": "01_acquisition_agreement.pdf", "expected": "By: James Mitchell, CEO", "summary": "TechCorp signs through CEO James Mitchell."}, + {"id": "acq_seller_signatory", "topic": "parties", "source": "01_acquisition_agreement.pdf", "expected": "By: Sarah Chen, Founder & CEO", "summary": "StartupXYZ signs through founder/CEO Sarah Chen."}, + {"id": "dd_preparer", "topic": "due_diligence", "source": "02_due_diligence_report.pdf", "expected": "Prepared by: Morrison & Associates, LLP", "summary": "Morrison & Associates prepared the due diligence report."}, + {"id": "dd_revenue", "topic": "financials", "source": "02_due_diligence_report.pdf", "expected": "Revenue for FY2024: $12.3 million (growth of 45% YoY)", "summary": "FY2024 revenue was $12.3M, up 45% YoY."}, + {"id": "dd_ebitda", "topic": "financials", "source": "02_due_diligence_report.pdf", "expected": "EBITDA: $2.1 million (17% margin)", "summary": "EBITDA was $2.1M at a 17% margin."}, + {"id": "dd_cash_position", "topic": "financials", "source": "02_due_diligence_report.pdf", "expected": "Cash position: $3.2 million as of November 30, 2024", "summary": "Cash was $3.2M as of 30 Nov 2024."}, + {"id": "dd_disclosed_debt", "topic": "financials", "source": "02_due_diligence_report.pdf", "expected": "Outstanding debt: $1.5 million", "summary": "Disclosed outstanding debt was $1.5M."}, + {"id": "dd_patent_count", "topic": "intellectual_property", "source": "02_due_diligence_report.pdf", "expected": "StartupXYZ holds 12 patents related to AI/ML technology", "summary": "Twelve AI/ML patents, per due diligence."}, + {"id": "dd_headcount", "topic": "employees", "source": "02_due_diligence_report.pdf", "expected": "Total employees: 47 (32 engineering, 8 sales, 7 operations)", "summary": "47 employees: 32 eng, 8 sales, 7 ops."}, + {"id": "dd_retention_risk", "topic": "employees", "source": "02_due_diligence_report.pdf", "expected": "Key employee retention risk: HIGH for 5 senior engineers", "summary": "Retention risk is HIGH for 5 senior engineers."}, + {"id": "dd_contracts_reviewed", "topic": "contracts", "source": "02_due_diligence_report.pdf", "expected": "23 active customer contracts reviewed", "summary": "23 active customer contracts were reviewed."}, + {"id": "dd_consent_contracts", "topic": "contracts", "source": "02_due_diligence_report.pdf", "expected": "3 contracts contain change-of-control provisions requiring consent", "summary": "Three contracts need change-of-control consent."}, + {"id": "dd_megacorp_concentration", "topic": "customer_concentration", "source": "02_due_diligence_report.pdf", "expected": "Largest customer (MegaCorp) accounts for 28% of revenue", "summary": "MegaCorp is 28% of revenue."}, + {"id": "dd_hsr_required", "topic": "regulatory", "source": "02_due_diligence_report.pdf", "expected": "HSR filing required - timeline in Document: Regulatory Approval Letter", "summary": "An HSR filing is required; timeline lives in the FTC letter."}, + {"id": "ip_certifier", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "From: PatentWatch Legal Services", "summary": "PatentWatch Legal Services issued the IP certification."}, + {"id": "ip_patent_count", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "StartupXYZ owns 12 U.S. patents as listed in Schedule 1 - IP Assets", "summary": "12 US patents, itemised in Schedule 1."}, + {"id": "ip_first_patent", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "US Patent 10,123,456", "summary": "One patent is US 10,123,456 (neural network optimization)."}, + {"id": "ip_encumbrances", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "All patents are valid, enforceable, and free of liens or encumbrances", "summary": "The patents are unencumbered."}, + {"id": "ip_trademark_count", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "StartupXYZ owns 3 registered trademarks", "summary": "Three registered trademarks."}, + {"id": "ip_product_mark", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "IntelliFlow", "summary": "IntelliFlow is the registered product name."}, + {"id": "ip_open_source", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "StartupXYZ uses 47 open-source libraries", "summary": "47 open-source libraries in use."}, + {"id": "ip_copyleft", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "no copyleft contamination issues identified", "summary": "No copyleft contamination was found."}, + {"id": "ip_pending_application", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "one pending patent application (Application No. 17/456,789)", "summary": "One pending application, 17/456,789."}, + {"id": "ip_pending_issue_date", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "expected to issue Q2 2025", "summary": "The pending application should issue in Q2 2025."}, + {"id": "ip_attorney", "topic": "intellectual_property", "source": "03_ip_certification.pdf", "expected": "By: Robert Kim, Patent Attorney", "summary": "Robert Kim signed the certification."}, + {"id": "risk_author", "topic": "risk", "source": "04_risk_assessment.pdf", "expected": "From: Corporate Development Team", "summary": "The risk memo comes from Corporate Development."}, + {"id": "risk_megacorp_impact", "topic": "customer_concentration", "source": "04_risk_assessment.pdf", "expected": "Impact if materialized: $3.4M annual revenue at risk", "summary": "$3.4M of annual revenue is at risk from MegaCorp."}, + {"id": "risk_megacorp_mitigation", "topic": "cross_reference", "source": "04_risk_assessment.pdf", "expected": "Mitigation: Obtain consent prior to closing (see Document: Customer Consent Letters)", "summary": "The concentration mitigation is a consent, tracked in the consent letters."}, + {"id": "risk_engineers_leaving", "topic": "employees", "source": "04_risk_assessment.pdf", "expected": "2 have expressed interest in leaving post-acquisition", "summary": "Two of the five key engineers may leave."}, + {"id": "risk_retention_cost", "topic": "employees", "source": "04_risk_assessment.pdf", "expected": "Estimated cost: $2.5M in retention bonuses", "summary": "Retention bonuses are estimated at $2.5M."}, + {"id": "risk_earnout_medium", "topic": "risk", "source": "04_risk_assessment.pdf", "expected": "Risk: Disagreement on metric calculation methodology", "summary": "The earnout's medium risk is metric-calculation disagreement."}, + {"id": "risk_integration_cost", "topic": "risk", "source": "04_risk_assessment.pdf", "expected": "Estimated integration costs: $4.2M over 18 months", "summary": "Integration is estimated at $4.2M over 18 months."}, + {"id": "risk_overrun_rate", "topic": "risk", "source": "04_risk_assessment.pdf", "expected": "Cost overruns of 20-30% typical in tech acquisitions", "summary": "20-30% cost overruns are typical."}, + {"id": "risk_total_impact", "topic": "risk", "source": "04_risk_assessment.pdf", "expected": "Total risk-adjusted impact: $6.2M - $8.7M", "summary": "Risk-adjusted impact is $6.2M-$8.7M."}, + {"id": "risk_patent_low", "topic": "risk", "source": "04_risk_assessment.pdf", "expected": "Low risk of rejection based on patent attorney's assessment", "summary": "The pending patent is a low risk."}, + {"id": "fin_wc_target", "topic": "price_adjustments", "source": "05_financial_adjustments.pdf", "expected": "Target working capital: $1,200,000", "summary": "Target working capital is $1.2M."}, + {"id": "fin_wc_closing", "topic": "price_adjustments", "source": "05_financial_adjustments.pdf", "expected": "Estimated closing working capital: $980,000", "summary": "Estimated closing working capital is $980K."}, + {"id": "fin_extra_debt", "topic": "price_adjustments", "source": "05_financial_adjustments.pdf", "expected": "Additional identified debt: $175,000 (capital lease obligations)", "summary": "$175K of extra debt from capital leases."}, + {"id": "fin_deferred_revenue", "topic": "price_adjustments", "source": "05_financial_adjustments.pdf", "expected": "Deferred revenue requiring restatement: $340,000", "summary": "$340K of deferred revenue needs restating."}, + {"id": "fin_multiple", "topic": "price_adjustments", "source": "05_financial_adjustments.pdf", "expected": "Implied value adjustment (at 15x)", "summary": "The revenue-recognition adjustment is valued at a 15x multiple."}, + {"id": "fin_reserve_total", "topic": "escrow", "source": "05_financial_adjustments.pdf", "expected": "Total reserve: $1,300,000", "summary": "The contingent-liability reserve is $1.3M."}, + {"id": "fin_concentration_reserve", "topic": "escrow", "source": "05_financial_adjustments.pdf", "expected": "Customer concentration risk: $500,000", "summary": "$500K is reserved against customer concentration."}, + {"id": "fin_adjusted_price", "topic": "purchase_price", "source": "05_financial_adjustments.pdf", "expected": "Adjusted Purchase Price: $43,330,000", "summary": "The adjusted purchase price is $43,330,000."}, + {"id": "fin_total_cash_required", "topic": "purchase_price", "source": "05_financial_adjustments.pdf", "expected": "Total Cash Required at Closing: $44,630,000", "summary": "The memo's total cash required at closing is $44,630,000."}, + {"id": "fin_revised_cash", "topic": "purchase_price", "source": "05_financial_adjustments.pdf", "expected": "Cash at closing: $28,330,000 (adjusted)", "summary": "Revised cash at closing is $28,330,000."}, + {"id": "fin_escrow_release", "topic": "escrow", "source": "05_financial_adjustments.pdf", "expected": "Escrow: $1,300,000 (18-month release schedule)", "summary": "The $1.3M escrow releases over 18 months."}, + {"id": "legal_opinion_date", "topic": "legal_opinion", "source": "06_legal_opinion.pdf", "expected": "Date: December 18, 2024", "summary": "The legal opinion is dated 18 Dec 2024."}, + {"id": "legal_counsel_role", "topic": "legal_opinion", "source": "06_legal_opinion.pdf", "expected": "We have acted as legal counsel to StartupXYZ LLC", "summary": "Wilson & Partners acted for the seller."}, + {"id": "legal_organisation_state", "topic": "legal_opinion", "source": "06_legal_opinion.pdf", "expected": "in good standing under the laws of Delaware", "summary": "StartupXYZ is a Delaware LLC in good standing."}, + {"id": "legal_no_conflicts_exception", "topic": "cross_reference", "source": "06_legal_opinion.pdf", "expected": "except for change-of-control provisions noted in Document: Customer Consent Letters", "summary": "The no-conflicts opinion carves out change-of-control consents."}, + {"id": "legal_no_litigation", "topic": "litigation", "source": "06_legal_opinion.pdf", "expected": "There is no litigation, arbitration, or governmental proceeding pending", "summary": "No litigation is pending or threatened."}, + {"id": "legal_tax_carveout", "topic": "legal_opinion", "source": "06_legal_opinion.pdf", "expected": "We express no opinion on tax matters", "summary": "Tax matters are excluded from the opinion."}, + {"id": "legal_jurisdiction_limit", "topic": "legal_opinion", "source": "06_legal_opinion.pdf", "expected": "This opinion is limited to Delaware and federal law", "summary": "The opinion covers only Delaware and federal law."}, + {"id": "legal_signatory", "topic": "legal_opinion", "source": "06_legal_opinion.pdf", "expected": "By: Jennifer Walsh, Partner", "summary": "Jennifer Walsh of Wilson & Partners signed it."}, + {"id": "nda_execution_date", "topic": "confidentiality", "source": "07_nda.pdf", "expected": "entered into as of October 1, 2024", "summary": "The mutual NDA is dated 1 Oct 2024."}, + {"id": "nda_seller_address", "topic": "parties", "source": "07_nda.pdf", "expected": "123 Innovation Way, Palo Alto, CA 94301", "summary": "StartupXYZ is at 123 Innovation Way, Palo Alto."}, + {"id": "nda_buyer_address", "topic": "parties", "source": "07_nda.pdf", "expected": "500 Technology Drive, San Francisco, CA 94105", "summary": "TechCorp is at 500 Technology Drive, San Francisco."}, + {"id": "nda_term", "topic": "confidentiality", "source": "07_nda.pdf", "expected": "remain in effect for three (3) years", "summary": "The NDA runs three years."}, + {"id": "nda_supersession", "topic": "cross_reference", "source": "07_nda.pdf", "expected": "until superseded by the confidentiality provisions in the Document: Acquisition Agreement", "summary": "The NDA is superseded by Article V of the Acquisition Agreement."}, + {"id": "nda_exclusion_public", "topic": "confidentiality", "source": "07_nda.pdf", "expected": "Is or becomes publicly available through no fault of the receiving Party", "summary": "Publicly available information is excluded."}, + {"id": "nda_exclusion_independent", "topic": "confidentiality", "source": "07_nda.pdf", "expected": "Is independently developed without use of Confidential Information", "summary": "Independently developed information is excluded."}, + {"id": "nda_return_of_materials", "topic": "confidentiality", "source": "07_nda.pdf", "expected": "each Party shall return or destroy all Confidential Information", "summary": "Materials must be returned or destroyed on request."}, + {"id": "nda_no_license", "topic": "cross_reference", "source": "07_nda.pdf", "expected": "Nothing in this NDA grants any rights to intellectual property", "summary": "The NDA grants no IP rights."}, + {"id": "reg_statute", "topic": "regulatory", "source": "08_regulatory_approval.pdf", "expected": "Hart-Scott-Rodino Antitrust Improvements Act of 1976", "summary": "The waiting period is the HSR Act's."}, + {"id": "reg_early_termination", "topic": "regulatory", "source": "08_regulatory_approval.pdf", "expected": "granted early termination of the waiting period", "summary": "The FTC granted early termination."}, + {"id": "reg_filing_date", "topic": "regulatory", "source": "08_regulatory_approval.pdf", "expected": "Filing Date: January 10, 2025", "summary": "The HSR filing went in on 10 Jan 2025."}, + {"id": "reg_filing_fee", "topic": "regulatory", "source": "08_regulatory_approval.pdf", "expected": "HSR Filing Fee: $30,000", "summary": "The HSR filing fee was $30,000."}, + {"id": "reg_termination_date", "topic": "regulatory", "source": "08_regulatory_approval.pdf", "expected": "Early Termination Granted: January 28, 2025", "summary": "Early termination was granted 28 Jan 2025."}, + {"id": "reg_condition_satisfied", "topic": "cross_reference", "source": "08_regulatory_approval.pdf", "expected": "satisfies the condition precedent set forth in Article IV, Section 4.1(a)", "summary": "Early termination satisfies AA Article IV 4.1(a)."}, + {"id": "reg_no_preclusion", "topic": "regulatory", "source": "08_regulatory_approval.pdf", "expected": "does not preclude the Commission from taking any action", "summary": "Early termination does not bar later FTC action."}, + {"id": "cons_report_date", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "Date: February 15, 2025", "summary": "The consent status report is dated 15 Feb 2025."}, + {"id": "cons_megacorp_status", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "MegaCorp Inc. - OBTAINED", "summary": "MegaCorp's consent was obtained."}, + {"id": "cons_megacorp_date", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "Consent Received: February 10, 2025", "summary": "MegaCorp consented on 10 Feb 2025."}, + {"id": "cons_megacorp_meeting", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "meeting held 2/8/25", "summary": "MegaCorp met TechCorp leadership on 8 Feb 2025."}, + {"id": "cons_dataflow_status", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "DataFlow Systems - OBTAINED", "summary": "DataFlow's consent was obtained."}, + {"id": "cons_dataflow_value", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "Contract Value: $1.2M annual", "summary": "DataFlow's contract is $1.2M a year."}, + {"id": "cons_cloudtech_status", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "CloudTech Partners - PENDING", "summary": "CloudTech's consent is still pending."}, + {"id": "cons_cloudtech_value", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "Contract Value: $890K annual", "summary": "CloudTech's contract is $890K a year."}, + {"id": "cons_cloudtech_expected", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "Expected: February 20, 2025", "summary": "CloudTech's consent is expected 20 Feb 2025."}, + {"id": "cons_obtained_revenue", "topic": "consents", "source": "09_customer_consents.pdf", "expected": "2 obtained (representing $4.6M annual revenue)", "summary": "Two consents cover $4.6M of annual revenue."}, + {"id": "close_date", "topic": "timeline", "source": "10_closing_checklist.pdf", "expected": "Closing Date: March 1, 2025", "summary": "Closing is set for 1 Mar 2025."}, + {"id": "close_location", "topic": "timeline", "source": "10_closing_checklist.pdf", "expected": "Closing Location: Wilson & Partners LLP, San Francisco", "summary": "Closing happens at Wilson & Partners in San Francisco."}, + {"id": "close_state_filings_open", "topic": "closing_status", "source": "10_closing_checklist.pdf", "expected": "[ ] State regulatory filings (if required)", "summary": "State regulatory filings are still unticked."}, + {"id": "close_cloudtech_open", "topic": "closing_status", "source": "10_closing_checklist.pdf", "expected": "[ ] CloudTech consent (expected February 20)", "summary": "The CloudTech consent box is still unticked."}, + {"id": "close_total_at_closing", "topic": "purchase_price", "source": "10_closing_checklist.pdf", "expected": "Total at Closing: $39,630,000", "summary": "The checklist totals closing funds at $39,630,000."}, + {"id": "close_cash_payment", "topic": "purchase_price", "source": "10_closing_checklist.pdf", "expected": "Cash payment: $28,330,000", "summary": "Cash payment at closing is $28,330,000."}, + {"id": "close_escrow_deposit", "topic": "escrow", "source": "10_closing_checklist.pdf", "expected": "Escrow deposit: $1,300,000", "summary": "The escrow deposit is $1,300,000."}, + {"id": "close_escrow_agent", "topic": "parties", "source": "10_closing_checklist.pdf", "expected": "Escrow Agent: First National Trust", "summary": "First National Trust is the escrow agent."}, + {"id": "close_buyer_counsel", "topic": "parties", "source": "10_closing_checklist.pdf", "expected": "Buyer's Counsel: Morrison & Associates LLP", "summary": "Morrison & Associates is the buyer's counsel."}, + {"id": "close_seller_counsel", "topic": "parties", "source": "10_closing_checklist.pdf", "expected": "Seller's Counsel: Wilson & Partners LLP", "summary": "Wilson & Partners is the seller's counsel."}, + {"id": "close_ip_assignment", "topic": "closing_status", "source": "10_closing_checklist.pdf", "expected": "[ ] IP Assignment Agreement (per Schedule 1 - IP Assets)", "summary": "The IP assignment is an outstanding closing document."}, + {"id": "close_ceo_contact", "topic": "parties", "source": "10_closing_checklist.pdf", "expected": "James Mitchell (CEO), (415) 555-0100", "summary": "TechCorp's contact is James Mitchell on (415) 555-0100."} + ], + "cross_references": [ + {"from": "01_acquisition_agreement.pdf", "to": "05_financial_adjustments.pdf", "kind": "exhibit", "cue": "Exhibit A - Financial Terms", "note": "Exhibit A is not a separate file; its detail lives in the Financial Adjustments Memo, which restates Exhibit A."}, + {"from": "01_acquisition_agreement.pdf", "to": "02_due_diligence_report.pdf", "kind": "document", "cue": "Document: Due Diligence Report"}, + {"from": "01_acquisition_agreement.pdf", "to": "03_ip_certification.pdf", "kind": "document", "cue": "Document: IP Certification Letter"}, + {"from": "01_acquisition_agreement.pdf", "to": "08_regulatory_approval.pdf", "kind": "document", "cue": "Document: Regulatory Approval"}, + {"from": "01_acquisition_agreement.pdf", "to": "04_risk_assessment.pdf", "kind": "document", "cue": "Document: Risk Assessment Memo"}, + {"from": "01_acquisition_agreement.pdf", "to": "07_nda.pdf", "kind": "document", "cue": "Document: Non-Disclosure Agreement"}, + {"from": "02_due_diligence_report.pdf", "to": "01_acquisition_agreement.pdf", "kind": "document", "cue": "Document: Acquisition Agreement"}, + {"from": "02_due_diligence_report.pdf", "to": "05_financial_adjustments.pdf", "kind": "document", "cue": "Document: Financial Adjustments Memo"}, + {"from": "02_due_diligence_report.pdf", "to": "03_ip_certification.pdf", "kind": "document", "cue": "Document: IP Certification Letter"}, + {"from": "02_due_diligence_report.pdf", "to": "06_legal_opinion.pdf", "kind": "document", "cue": "Document: Legal Opinion Letter"}, + {"from": "02_due_diligence_report.pdf", "to": "04_risk_assessment.pdf", "kind": "document", "cue": "Document: Risk Assessment Memo"}, + {"from": "02_due_diligence_report.pdf", "to": "08_regulatory_approval.pdf", "kind": "document", "cue": "Document: Regulatory Approval Letter"}, + {"from": "03_ip_certification.pdf", "to": "01_acquisition_agreement.pdf", "kind": "document", "cue": "Document: Acquisition Agreement"}, + {"from": "03_ip_certification.pdf", "to": "07_nda.pdf", "kind": "document", "cue": "Document: Non-Disclosure Agreement template"}, + {"from": "03_ip_certification.pdf", "to": "04_risk_assessment.pdf", "kind": "document", "cue": "Document: Risk Assessment Memo"}, + {"from": "03_ip_certification.pdf", "to": "06_legal_opinion.pdf", "kind": "document", "cue": "Document: Legal Opinion Letter"}, + {"from": "04_risk_assessment.pdf", "to": "01_acquisition_agreement.pdf", "kind": "document", "cue": "Document: Acquisition Agreement"}, + {"from": "04_risk_assessment.pdf", "to": "09_customer_consents.pdf", "kind": "document", "cue": "Document: Customer Consent Letters"}, + {"from": "04_risk_assessment.pdf", "to": "03_ip_certification.pdf", "kind": "document", "cue": "Document: IP Certification Letter"}, + {"from": "04_risk_assessment.pdf", "to": "08_regulatory_approval.pdf", "kind": "document", "cue": "Document: Regulatory Approval Letter"}, + {"from": "04_risk_assessment.pdf", "to": "05_financial_adjustments.pdf", "kind": "document", "cue": "Document: Financial Adjustments Memo"}, + {"from": "04_risk_assessment.pdf", "to": "02_due_diligence_report.pdf", "kind": "document", "cue": "Document: Due Diligence Report"}, + {"from": "04_risk_assessment.pdf", "to": "10_closing_checklist.pdf", "kind": "document", "cue": "Document: Closing Checklist"}, + {"from": "04_risk_assessment.pdf", "to": null, "kind": "document", "cue": "Document: Integration Plan", "dangling": true}, + {"from": "05_financial_adjustments.pdf", "to": "02_due_diligence_report.pdf", "kind": "document", "cue": "Document: Due Diligence Report"}, + {"from": "05_financial_adjustments.pdf", "to": "01_acquisition_agreement.pdf", "kind": "document", "cue": "Document: Acquisition Agreement"}, + {"from": "05_financial_adjustments.pdf", "to": "04_risk_assessment.pdf", "kind": "document", "cue": "Document: Risk Assessment Memo"}, + {"from": "05_financial_adjustments.pdf", "to": "10_closing_checklist.pdf", "kind": "document", "cue": "Document: Closing Checklist"}, + {"from": "06_legal_opinion.pdf", "to": "01_acquisition_agreement.pdf", "kind": "document", "cue": "Document: Acquisition Agreement"}, + {"from": "06_legal_opinion.pdf", "to": "02_due_diligence_report.pdf", "kind": "document", "cue": "Document: Due Diligence Report"}, + {"from": "06_legal_opinion.pdf", "to": "03_ip_certification.pdf", "kind": "document", "cue": "Document: IP Certification Letter"}, + {"from": "06_legal_opinion.pdf", "to": "07_nda.pdf", "kind": "document", "cue": "Document: Non-Disclosure Agreement"}, + {"from": "06_legal_opinion.pdf", "to": "09_customer_consents.pdf", "kind": "document", "cue": "Document: Customer Consent Letters"}, + {"from": "06_legal_opinion.pdf", "to": "08_regulatory_approval.pdf", "kind": "document", "cue": "Document: Regulatory Approval Letter"}, + {"from": "07_nda.pdf", "to": "01_acquisition_agreement.pdf", "kind": "document", "cue": "Document: Acquisition Agreement"}, + {"from": "07_nda.pdf", "to": "02_due_diligence_report.pdf", "kind": "document", "cue": "Document: Due Diligence Report"}, + {"from": "07_nda.pdf", "to": "03_ip_certification.pdf", "kind": "document", "cue": "Document: IP Certification Letter"}, + {"from": "08_regulatory_approval.pdf", "to": "01_acquisition_agreement.pdf", "kind": "document", "cue": "Document: Acquisition Agreement"}, + {"from": "08_regulatory_approval.pdf", "to": "10_closing_checklist.pdf", "kind": "document", "cue": "Document: Closing Checklist"}, + {"from": "08_regulatory_approval.pdf", "to": "04_risk_assessment.pdf", "kind": "document", "cue": "Document: Risk Assessment Memo"}, + {"from": "08_regulatory_approval.pdf", "to": "02_due_diligence_report.pdf", "kind": "document", "cue": "Document: Due Diligence Report"}, + {"from": "09_customer_consents.pdf", "to": "01_acquisition_agreement.pdf", "kind": "document", "cue": "Document: Acquisition Agreement"}, + {"from": "09_customer_consents.pdf", "to": "04_risk_assessment.pdf", "kind": "document", "cue": "Document: Risk Assessment Memo"}, + {"from": "09_customer_consents.pdf", "to": "02_due_diligence_report.pdf", "kind": "document", "cue": "Document: Due Diligence Report"}, + {"from": "09_customer_consents.pdf", "to": "10_closing_checklist.pdf", "kind": "document", "cue": "Document: Closing Checklist"}, + {"from": "10_closing_checklist.pdf", "to": "08_regulatory_approval.pdf", "kind": "document", "cue": "Document: Regulatory Approval Letter"}, + {"from": "10_closing_checklist.pdf", "to": "09_customer_consents.pdf", "kind": "document", "cue": "Document: Customer Consent Letters"}, + {"from": "10_closing_checklist.pdf", "to": "02_due_diligence_report.pdf", "kind": "document", "cue": "Document: Due Diligence Report"}, + {"from": "10_closing_checklist.pdf", "to": "06_legal_opinion.pdf", "kind": "document", "cue": "Document: Legal Opinion Letter"}, + {"from": "10_closing_checklist.pdf", "to": "03_ip_certification.pdf", "kind": "document", "cue": "Document: IP Certification Letter"}, + {"from": "10_closing_checklist.pdf", "to": "04_risk_assessment.pdf", "kind": "document", "cue": "Document: Risk Assessment Memo"}, + {"from": "10_closing_checklist.pdf", "to": "01_acquisition_agreement.pdf", "kind": "document", "cue": "Document: Acquisition Agreement"}, + {"from": "10_closing_checklist.pdf", "to": "05_financial_adjustments.pdf", "kind": "document", "cue": "Document: Financial Adjustments Memo"}, + {"from": "10_closing_checklist.pdf", "to": null, "kind": "document", "cue": "Document: Integration Plan", "dangling": true} + ] +} diff --git a/eval/corpora/acquisition/01_acquisition_agreement.pdf b/eval/corpora/acquisition/01_acquisition_agreement.pdf new file mode 100644 index 00000000..a8f1c469 Binary files /dev/null and b/eval/corpora/acquisition/01_acquisition_agreement.pdf differ diff --git a/eval/corpora/acquisition/02_due_diligence_report.pdf b/eval/corpora/acquisition/02_due_diligence_report.pdf new file mode 100644 index 00000000..98ff0cb5 Binary files /dev/null and b/eval/corpora/acquisition/02_due_diligence_report.pdf differ diff --git a/eval/corpora/acquisition/03_ip_certification.pdf b/eval/corpora/acquisition/03_ip_certification.pdf new file mode 100644 index 00000000..b2bb9415 Binary files /dev/null and b/eval/corpora/acquisition/03_ip_certification.pdf differ diff --git a/eval/corpora/acquisition/04_risk_assessment.pdf b/eval/corpora/acquisition/04_risk_assessment.pdf new file mode 100644 index 00000000..773335bc Binary files /dev/null and b/eval/corpora/acquisition/04_risk_assessment.pdf differ diff --git a/eval/corpora/acquisition/05_financial_adjustments.pdf b/eval/corpora/acquisition/05_financial_adjustments.pdf new file mode 100644 index 00000000..404f3948 Binary files /dev/null and b/eval/corpora/acquisition/05_financial_adjustments.pdf differ diff --git a/eval/corpora/acquisition/06_legal_opinion.pdf b/eval/corpora/acquisition/06_legal_opinion.pdf new file mode 100644 index 00000000..37e4549f Binary files /dev/null and b/eval/corpora/acquisition/06_legal_opinion.pdf differ diff --git a/eval/corpora/acquisition/07_nda.pdf b/eval/corpora/acquisition/07_nda.pdf new file mode 100644 index 00000000..4efe2bff Binary files /dev/null and b/eval/corpora/acquisition/07_nda.pdf differ diff --git a/eval/corpora/acquisition/08_regulatory_approval.pdf b/eval/corpora/acquisition/08_regulatory_approval.pdf new file mode 100644 index 00000000..44194d64 Binary files /dev/null and b/eval/corpora/acquisition/08_regulatory_approval.pdf differ diff --git a/eval/corpora/acquisition/09_customer_consents.pdf b/eval/corpora/acquisition/09_customer_consents.pdf new file mode 100644 index 00000000..597a1c84 Binary files /dev/null and b/eval/corpora/acquisition/09_customer_consents.pdf differ diff --git a/eval/corpora/acquisition/10_closing_checklist.pdf b/eval/corpora/acquisition/10_closing_checklist.pdf new file mode 100644 index 00000000..6d663d3b Binary files /dev/null and b/eval/corpora/acquisition/10_closing_checklist.pdf differ diff --git a/eval/corpora/atlas7_service_manual.facts.json b/eval/corpora/atlas7_service_manual.facts.json new file mode 100644 index 00000000..29f4db60 --- /dev/null +++ b/eval/corpora/atlas7_service_manual.facts.json @@ -0,0 +1,29 @@ +{ + "corpus": "atlas7", + "document": "atlas7_service_manual.pdf", + "generator": "hand-authored PDF (planted-fact test document, 2026-08-08)", + "note": "Fictional machine and fictional manufacturer. Every 'expected' string is a verbatim substring of the PDF text after whitespace normalisation (checked by eval/corpora/verify_facts.py).", + "facts": [ + {"id": "atlas_model_revision", "topic": "identification", "expected": "Atlas-7 Dual Boiler (2026 revision C)", "summary": "The model is the Atlas-7 Dual Boiler, 2026 revision C."}, + {"id": "atlas_manufacturer", "topic": "identification", "expected": "Meridian Coffee Systems, Tacoma WA", "summary": "Made by Meridian Coffee Systems of Tacoma, WA."}, + {"id": "atlas_brew_pressure", "topic": "specifications", "expected": "pressure of 9.2 bar", "summary": "The brew boiler runs at 9.2 bar during extraction."}, + {"id": "atlas_steam_pressure", "topic": "specifications", "expected": "steam boiler is maintained at 1.45 bar", "summary": "The steam boiler is held at 1.45 bar."}, + {"id": "atlas_brew_temperature", "topic": "specifications", "expected": "93.5 degrees Celsius", "summary": "The PID holds brew water at 93.5 C."}, + {"id": "atlas_temperature_tolerance", "topic": "specifications", "expected": "tolerance of 0.4 degrees", "summary": "Brew temperature tolerance is 0.4 degrees."}, + {"id": "atlas_pump_rating", "topic": "specifications", "expected": "52 watts continuous duty", "summary": "The vibratory pump is rated 52 W continuous."}, + {"id": "atlas_descale_interval", "topic": "maintenance", "expected": "every 60 days when water hardness exceeds", "summary": "Descale every 60 days above the hardness threshold."}, + {"id": "atlas_water_hardness", "topic": "maintenance", "expected": "120 ppm", "summary": "The hardness threshold is 120 ppm."}, + {"id": "atlas_gasket_part", "topic": "maintenance", "expected": "group head gasket (part MG-311)", "summary": "The group head gasket is part MG-311."}, + {"id": "atlas_gasket_interval", "topic": "maintenance", "expected": "replaced every 14 months", "summary": "The gasket is replaced every 14 months."}, + {"id": "atlas_backflush", "topic": "maintenance", "expected": "Backflushing with Cafiza detergent is recommended weekly", "summary": "Weekly backflush with Cafiza."}, + {"id": "atlas_e11", "topic": "error_codes", "expected": "E11: Brew boiler thermistor open circuit", "summary": "E11 is a brew boiler thermistor open circuit."}, + {"id": "atlas_e11_part", "topic": "error_codes", "expected": "Replace sensor part TS-71", "summary": "E11 is fixed by replacing sensor TS-71."}, + {"id": "atlas_e23", "topic": "error_codes", "expected": "Check the OPV calibration at 12 bar", "summary": "E23 is steam overpressure; check OPV calibration at 12 bar."}, + {"id": "atlas_e42_prime", "topic": "error_codes", "expected": "Prime the pump by running 200 ml", "summary": "E42 is pump cavitation; prime with 200 ml."}, + {"id": "atlas_e42_procedure", "topic": "error_codes", "expected": "hot water wand, then power cycle the unit", "summary": "Prime via the hot water wand, then power cycle."}, + {"id": "atlas_e57", "topic": "error_codes", "expected": "Clean the inlet mesh filter", "summary": "E57 is zero flow-meter pulses; clean the inlet mesh filter."}, + {"id": "atlas_warranty_length", "topic": "warranty", "expected": "36-month parts warranty", "summary": "The parts warranty is 36 months."}, + {"id": "atlas_warranty_void", "topic": "warranty", "expected": "citric acid above 8 percent concentration", "summary": "Third-party descalers over 8 percent citric acid void the warranty."}, + {"id": "atlas_serial_location", "topic": "warranty", "expected": "engraved under the drip tray on the left rail", "summary": "The serial number is under the drip tray on the left rail."} + ] +} diff --git a/eval/corpora/atlas7_service_manual.pdf b/eval/corpora/atlas7_service_manual.pdf new file mode 100644 index 00000000..3987ddb1 Binary files /dev/null and b/eval/corpora/atlas7_service_manual.pdf differ diff --git a/eval/corpora/make_hr_handbook.py b/eval/corpora/make_hr_handbook.py new file mode 100644 index 00000000..c7c7a348 --- /dev/null +++ b/eval/corpora/make_hr_handbook.py @@ -0,0 +1,124 @@ +"""Generate the synthetic HR leave-policy corpus used by the eval harness. + +The document is entirely fictional (Northwind Robotics is not a real company) +so that no planted fact can be answered from the generation model's parametric +knowledge โ€” a wrong answer is unambiguously a retrieval or grounding failure. + +Run from the repo root: + + .venv/bin/python eval/corpora/make_hr_handbook.py + +Writes ``eval/corpora/northwind_leave_policy.pdf``. The planted facts live in +``eval/corpora/northwind_leave_policy.facts.json`` (hand-maintained sidecar); +this script asserts every ``expected`` substring in that sidecar really appears +in the rendered page text before it saves the file. +""" + +import json +import os +import sys + +import pymupdf + +HERE = os.path.dirname(os.path.abspath(__file__)) +PDF_PATH = os.path.join(HERE, "northwind_leave_policy.pdf") +FACTS_PATH = os.path.join(HERE, "northwind_leave_policy.facts.json") + +PAGES = [ + ( + "Northwind Robotics - Leave and Absence Policy Handbook", + """Policy PPL-204, revision 4. Effective 1 February 2026. +Owner: Department of People Operations, Northwind Robotics, Gothenburg. + +1. ANNUAL LEAVE +Employees below Grade 7 accrue 23 days of paid annual leave per calendar +year. Employees at Grade 7 and above accrue 28 days. Accrual begins on the +first day of employment and is credited monthly in arrears. + +A maximum of 5 unused annual leave days may be carried into the following +year. Carried days expire on 31 March and are not paid out on expiry. + +2. REQUESTING LEAVE +All leave requests are submitted through the Kestrel HR portal at least 10 +working days before the intended start date. Any absence longer than 10 +consecutive working days additionally requires written approval from a +director. Requests are answered within 3 working days. +""", + ), + ( + "Northwind Robotics - Sickness, Parental and Special Leave", + """3. SICK LEAVE +Sick leave is paid at 100 percent of base salary for the first 12 weeks of a +single absence, and at 60 percent for a further 8 weeks. A medical +certificate is required once an absence exceeds 4 consecutive working days. + +4. PARENTAL LEAVE +Parental leave is 18 weeks per child, of which 6 weeks are fully paid. It +must be taken before the child's third birthday. Parental leave may be split +into no more than 3 separate blocks. + +5. BEREAVEMENT LEAVE +Bereavement leave is 5 working days for an immediate family member and 2 +working days for an extended family member. + +6. JURY DUTY +Jury service is paid in full for up to 15 working days per calendar year. +Any court allowance received must be surrendered to the payroll team. +""", + ), + ( + "Northwind Robotics - Sabbatical, Holidays and Exclusions", + """7. UNPAID SABBATICAL +Employees with at least 4 years of continuous service may apply for an +unpaid sabbatical of up to 90 days. Applications require 60 days written +notice and are approved by the Head of People Operations. A sabbatical does +not interrupt continuous-service accrual. + +8. PUBLIC HOLIDAYS +Northwind Robotics recognises 9 public holidays. An employee rostered to +work on a public holiday is paid at 1.5 times the normal rate and receives a +substitute day off within the same quarter. + +9. EXCLUSIONS AND FORFEITURE +Annual leave is not paid out on resignation unless the employee has served +more than 6 months. Leave taken without portal approval is recorded as +unauthorised absence and is unpaid. Contractors engaged through an agency are +not covered by this policy. +""", + ), +] + + +def build() -> None: + doc = pymupdf.open() + for title, body in PAGES: + page = doc.new_page() + page.insert_text((60, 70), title, fontsize=13, fontname="helv") + text_y = 100 + for line in body.strip().split("\n"): + page.insert_text((60, text_y), line, fontsize=10, fontname="helv") + text_y += 15 + doc.save(PDF_PATH) + doc.close() + + +def verify() -> int: + doc = pymupdf.open(PDF_PATH) + full_text = " ".join(page.get_text() for page in doc) + doc.close() + normalised = " ".join(full_text.split()) + + with open(FACTS_PATH, "r", encoding="utf-8") as fh: + facts = json.load(fh)["facts"] + + missing = [f["id"] for f in facts if f["expected"] not in normalised] + if missing: + print(f"MISSING expected substrings in rendered PDF: {missing}") + return 1 + print(f"OK: {len(facts)} planted facts all present in {PDF_PATH}") + return 0 + + +if __name__ == "__main__": + build() + raise SystemExit(verify()) diff --git a/eval/corpora/northwind_leave_policy.facts.json b/eval/corpora/northwind_leave_policy.facts.json new file mode 100644 index 00000000..b2854f03 --- /dev/null +++ b/eval/corpora/northwind_leave_policy.facts.json @@ -0,0 +1,32 @@ +{ + "corpus": "hr", + "document": "northwind_leave_policy.pdf", + "generator": "eval/corpora/make_hr_handbook.py", + "note": "Fictional company and fictional policy numbers. Every 'expected' string below is a verbatim substring of the rendered PDF text after whitespace normalisation (make_hr_handbook.py asserts this on every build).", + "facts": [ + {"id": "hr_annual_below_g7", "topic": "annual_leave", "expected": "23 days of paid annual leave", "summary": "Employees below Grade 7 accrue 23 days of paid annual leave per calendar year."}, + {"id": "hr_annual_g7_plus", "topic": "annual_leave", "expected": "Grade 7 and above accrue 28 days", "summary": "Grade 7 and above accrue 28 days."}, + {"id": "hr_carryover_cap", "topic": "annual_leave", "expected": "maximum of 5 unused annual leave days may be carried", "summary": "At most 5 unused annual leave days carry into the next year."}, + {"id": "hr_carryover_expiry", "topic": "annual_leave", "expected": "Carried days expire on 31 March", "summary": "Carried days expire on 31 March and are not paid out."}, + {"id": "hr_request_notice", "topic": "requesting_leave", "expected": "Kestrel HR portal at least 10", "summary": "Requests go through the Kestrel HR portal at least 10 working days ahead."}, + {"id": "hr_director_approval", "topic": "requesting_leave", "expected": "consecutive working days additionally requires written approval", "summary": "Absences over 10 consecutive working days need written director approval."}, + {"id": "hr_sick_full_pay", "topic": "sick_leave", "expected": "100 percent of base salary for the first 12 weeks", "summary": "Sick leave is paid at 100 percent for the first 12 weeks."}, + {"id": "hr_sick_reduced_pay", "topic": "sick_leave", "expected": "at 60 percent for a further 8 weeks", "summary": "Then 60 percent for a further 8 weeks."}, + {"id": "hr_medical_certificate", "topic": "sick_leave", "expected": "exceeds 4 consecutive working days", "summary": "A medical certificate is required past 4 consecutive working days."}, + {"id": "hr_parental_length", "topic": "parental_leave", "expected": "Parental leave is 18 weeks per child, of which 6 weeks are fully paid", "summary": "18 weeks per child, 6 of them fully paid."}, + {"id": "hr_parental_deadline", "topic": "parental_leave", "expected": "before the child's third birthday", "summary": "Parental leave must be taken before the child's third birthday."}, + {"id": "hr_parental_blocks", "topic": "parental_leave", "expected": "no more than 3 separate blocks", "summary": "Parental leave may be split into at most 3 blocks."}, + {"id": "hr_bereavement", "topic": "bereavement", "expected": "5 working days for an immediate family member", "summary": "5 days immediate family, 2 days extended family."}, + {"id": "hr_jury_duty", "topic": "jury_duty", "expected": "paid in full for up to 15 working days", "summary": "Jury service is paid in full up to 15 working days per year."}, + {"id": "hr_sabbatical_eligibility", "topic": "sabbatical", "expected": "at least 4 years of continuous service", "summary": "Sabbatical eligibility requires 4 years of continuous service."}, + {"id": "hr_sabbatical_length", "topic": "sabbatical", "expected": "unpaid sabbatical of up to 90 days", "summary": "Unpaid sabbatical of up to 90 days."}, + {"id": "hr_sabbatical_notice", "topic": "sabbatical", "expected": "require 60 days written", "summary": "Sabbatical applications require 60 days written notice."}, + {"id": "hr_sabbatical_approver", "topic": "sabbatical", "expected": "approved by the Head of People Operations", "summary": "Approved by the Head of People Operations."}, + {"id": "hr_public_holiday_count", "topic": "public_holidays", "expected": "recognises 9 public holidays", "summary": "Nine recognised public holidays."}, + {"id": "hr_public_holiday_pay", "topic": "public_holidays", "expected": "paid at 1.5 times the normal rate", "summary": "Working a public holiday pays 1.5x plus a substitute day."}, + {"id": "hr_resignation_payout", "topic": "exclusions", "expected": "not paid out on resignation", "summary": "Annual leave is not paid out on resignation under 6 months of service."}, + {"id": "hr_contractors_excluded", "topic": "exclusions", "expected": "Contractors engaged through an agency", "summary": "Agency contractors are not covered by the policy."}, + {"id": "hr_policy_id", "topic": "policy_metadata", "expected": "Policy PPL-204, revision 4", "summary": "The policy is PPL-204 revision 4, effective 1 February 2026."}, + {"id": "hr_policy_owner", "topic": "policy_metadata", "expected": "Department of People Operations, Northwind Robotics, Gothenburg", "summary": "Owned by People Operations in Gothenburg."} + ] +} diff --git a/eval/corpora/northwind_leave_policy.pdf b/eval/corpora/northwind_leave_policy.pdf new file mode 100644 index 00000000..61624564 Binary files /dev/null and b/eval/corpora/northwind_leave_policy.pdf differ diff --git a/eval/corpora/repo_docs.facts.json b/eval/corpora/repo_docs.facts.json new file mode 100644 index 00000000..cad8ccac --- /dev/null +++ b/eval/corpora/repo_docs.facts.json @@ -0,0 +1,226 @@ +{ + "corpus": "docs", + "document": "Documentation/*.md (referenced in place, not copied)", + "generator": "hand-selected prose anchors", + "note": "This corpus is the repo's own documentation. Anchors are verbatim prose substrings of the named source file after whitespace normalisation (checked by eval/corpora/verify_facts.py). Prose was preferred over table cells and fenced code because the docling markdown converter may reflow those.", + "source_glob": "Documentation/*.md", + "facts": [ + { + "id": "docs_no_weighted_blend", + "topic": "hybrid_retrieval", + "source": "retrieval_pipeline.md", + "expected": "There is no weighted linear blend", + "summary": "Hybrid fusion is pure RRF; there is no weighted linear blend and no dense_weight knob." + }, + { + "id": "docs_query_embed_cache", + "topic": "hybrid_retrieval", + "source": "retrieval_pipeline.md", + "expected": "memoised in a 256-entry", + "summary": "The query embedding is memoised in a 256-entry lru_cache per retriever." + }, + { + "id": "docs_api_single_threaded", + "topic": "operations", + "source": "retrieval_pipeline.md", + "expected": "so requests are handled one at a time", + "summary": "The RAG API is a single-threaded TCPServer, so requests are handled one at a time." + }, + { + "id": "docs_brute_force_vector", + "topic": "vector_index", + "source": "retrieval_pipeline.md", + "expected": "Vector search is a brute-force scan", + "summary": "No ANN index is built, so vector search is an exhaustive scan." + }, + { + "id": "docs_reindex_required", + "topic": "embedding_model", + "source": "retrieval_pipeline.md", + "expected": "Changing the embedding model requires re-indexing", + "summary": "Changing the embedding model requires re-indexing." + }, + { + "id": "docs_no_citation_markers", + "topic": "synthesis", + "source": "retrieval_pipeline.md", + "expected": "There are no inline citation markers", + "summary": "There are no inline citation markers; sources come back as source_documents." + }, + { + "id": "docs_expansion_filtered_out", + "topic": "context_expansion", + "source": "retrieval_pipeline.md", + "expected": "the freshly added neighbours are filtered back out", + "summary": "When the reranker ran, context-expansion neighbours are filtered back out." + }, + { + "id": "docs_provence_model", + "topic": "pruning", + "source": "retrieval_pipeline.md", + "expected": "naver/provence-reranker-debertav3-v1", + "summary": "Sentence pruning uses naver/provence-reranker-debertav3-v1." + }, + { + "id": "docs_pruning_off_by_default", + "topic": "pruning", + "source": "retrieval_pipeline.md", + "expected": "so pruning is off unless a request enables it", + "summary": "No shipped profile has a provence block, so pruning is off by default." + }, + { + "id": "docs_verifier_context_clamp", + "topic": "verifier", + "source": "verifier.md", + "expected": "clamped to the first 4000 characters", + "summary": "The verifier prompt clamps context to the first 4000 characters." + }, + { + "id": "docs_verifier_one_call_site", + "topic": "verifier", + "source": "verifier.md", + "expected": "There is exactly one call site in the repository", + "summary": "The verifier has exactly one call site." + }, + { + "id": "docs_verifier_cost", + "topic": "verifier", + "source": "verifier.md", + "expected": "one extra LLM round-trip per answered query", + "summary": "Verification costs one extra utility-model round-trip per answered query." + }, + { + "id": "docs_verifier_zero_score", + "topic": "verifier", + "source": "verifier.md", + "expected": "0 is treated as a parse failure", + "summary": "A confidence score of 0 is treated as a parse failure and no tag is appended." + }, + { + "id": "docs_triage_no_regex", + "topic": "triage", + "source": "triage_system.md", + "expected": "There is no regex or keyword stage in the agent", + "summary": "The agent router has no regex or keyword stage." + }, + { + "id": "docs_triage_overview_cap", + "topic": "triage", + "source": "triage_system.md", + "expected": "the first 40 loaded overviews", + "summary": "The overview router uses the first 40 loaded overviews." + }, + { + "id": "docs_triage_no_switch", + "topic": "triage", + "source": "triage_system.md", + "expected": "There is no global triage on/off switch", + "summary": "There is no global triage on/off switch and no similarity threshold." + }, + { + "id": "docs_triage_utility_model", + "topic": "triage", + "source": "triage_system.md", + "expected": "costs an LLM call, on the utility model", + "summary": "Only the agent-side router makes an LLM call (utility model qwen3.5:4b); the gateway gate is pure Python." + }, + { + "id": "docs_chunk_size_layering", + "topic": "chunking", + "source": "indexing_pipeline.md", + "expected": "while the HTTP path always sends", + "summary": "The CLI path uses the 1500-token code default while the HTTP path always sends 512." + }, + { + "id": "docs_txt_bypasses_docling", + "topic": "conversion", + "source": "indexing_pipeline.md", + "expected": "files bypass docling entirely and are wrapped in a fenced code block", + "summary": ".txt files bypass docling and are wrapped in a fenced code block." + }, + { + "id": "docs_no_overlap_knob", + "topic": "chunking", + "source": "indexing_pipeline.md", + "expected": "the legacy path has no overlap logic at all", + "summary": "There is no chunk-overlap knob; the legacy chunker has no overlap logic." + }, + { + "id": "docs_embedding_dimensions", + "topic": "embedding_model", + "source": "indexing_pipeline.md", + "expected": "produces 2560-dim vectors", + "summary": "Qwen3-Embedding-4B is 2560-dim, the 0.6B is 1024-dim; not interchangeable." + }, + { + "id": "docs_no_ann_index", + "topic": "vector_index", + "source": "indexing_pipeline.md", + "expected": "No ANN index is created", + "summary": "No ANN/IVF-PQ index is created anywhere in rag_system." + }, + { + "id": "docs_indexing_sequential", + "topic": "enrichment", + "source": "indexing_pipeline.md", + "expected": "there is no thread or process pool anywhere in the indexing path", + "summary": "Indexing is sequential; batch size controls reporting and memory, not concurrency." + }, + { + "id": "docs_enrichment_short_summary", + "topic": "enrichment", + "source": "indexing_pipeline.md", + "expected": "a summary shorter than 5 characters is discarded", + "summary": "An enrichment summary under 5 characters is discarded and the chunk is indexed unenriched." + }, + { + "id": "docs_overview_truncation", + "topic": "overviews", + "source": "indexing_pipeline.md", + "expected": "truncated to 5000 characters", + "summary": "Overview input is the first N chunks truncated to 5000 characters." + }, + { + "id": "docs_latechunk_cost", + "topic": "late_chunking", + "source": "indexing_pipeline.md", + "expected": "roughly double the vectors written", + "summary": "Late chunking costs a second copy of the embedder and roughly double the vectors." + }, + { + "id": "docs_identity_marker", + "topic": "index_safety", + "source": "system_overview.md", + "expected": "records the embedding model that wrote it", + "summary": "Each table records its embedding model + normalized flag; mismatch raises EmbedderMismatchError at index and query time." + }, + { + "id": "docs_dimension_mismatch_raises", + "topic": "vector_index", + "source": "indexing_pipeline.md", + "expected": "Silently dropping or recreating an index would corrupt it", + "summary": "A dimension mismatch is a hard ValueError on purpose." + }, + { + "id": "docs_no_hardcoded_model_in_prompt", + "topic": "prompts", + "source": "prompt_inventory.md", + "expected": "No prompt hard-codes a model name", + "summary": "No prompt hard-codes a model name; roles resolve through OLLAMA_CONFIG." + }, + { + "id": "docs_legacy_example_blocks", + "topic": "prompts", + "source": "prompt_inventory.md", + "expected": "Two legacy example blocks still live in the file", + "summary": "Two legacy decomposition example blocks are present but excluded from the assembled prompt." + }, + { + "id": "docs_direct_answer_length", + "topic": "prompts", + "source": "prompt_inventory.md", + "expected": "Caps the reply at 1-2 sentences", + "summary": "The direct_answer prompt caps the reply at 1-2 sentences." + } + ] +} \ No newline at end of file diff --git a/eval/corpora/rfc/MANIFEST.md b/eval/corpora/rfc/MANIFEST.md new file mode 100644 index 00000000..55e7cbfd --- /dev/null +++ b/eval/corpora/rfc/MANIFEST.md @@ -0,0 +1,160 @@ +# `rfc` โ€” 23 interlinked IETF RFCs (the QUIC / HTTP-3 family) + +**Real, third-party, plain-text documents the pipeline has never seen.** Every +other corpus in `eval/corpora/` is either synthetic (`atlas7`, `hr`, +`acquisition`) or this project's own writing (`docs`). This one is neither: the +files are byte-for-byte what the RFC Editor serves, written by people who never +heard of localGPT, and their naming and cross-referencing conventions are +therefore *not ones we invented*. That is the whole point of the corpus โ€” it is +the first honest test of the index-time cross-reference extractor +(`rag_system/indexing/crossref.py`) and of the chunker on documents whose +structure we did not choose. + +* **23 documents, 1,511,267 bytes (1.44 MiB).** +* Source: `https://www.rfc-editor.org/rfc/rfcNNNN.txt` โ€” the canonical + plain-text rendering. Nothing is edited after download. +* Reproduce with `.venv/bin/python eval/corpora/rfc/download.py` + (`--check` re-verifies sizes and the link graph without downloading). +* Answer-bearing anchors: `rfc.facts.json` (26 facts). Gold set: + `eval/goldset/rfc.jsonl` (24 rows). Row-level gate: + `.venv/bin/python eval/verify_rfc_goldset.py`. + +## Selection rule + +The cluster is the QUIC / HTTP-3 protocol family plus the documents it is +defined against. **Every file references, or is referenced by, at least two +others in the set** โ€” checked mechanically by `download.py --check`, which +counts `RFC NNNN` / `[RFCNNNN]` mentions across the corpus. There are **110 +directed intra-corpus references**; the lowest-degree document touches 4 others. + +Three sub-clusters, which is what makes the cross-reference gold rows possible: + +1. **QUIC core** (8999, 9000, 9001, 9002, 9221, 9369, 9308, 9312) โ€” the + transport, its TLS binding, its loss recovery, its version invariants. These + defer to each other constantly: 9000 hands the Retry Integrity Tag to 9001 + and the probe timeout to 9002; 9002 hands the anti-amplification limit back + to 9000; 9369 redefines 9001's key-derivation constants. +2. **HTTP over QUIC** (9114, 9204, 9218, 9220, 9297, 9298, 9412) plus their + **HTTP/2 counterparts** (8336, 8441). Each HTTP/3 document is deliberately + thin and defers its semantics to the HTTP/2 document it mirrors: 9220 reuses + 8441's `:protocol` pseudo-header, 9412 reuses 8336's ORIGIN payload. +3. **Shared normative dependencies** (2119, 8174, 8126, 6066, 7301) โ€” BCP 14, + the IANA registration policies, and the two TLS extensions the QUIC + documents build on. + +## Files + +`->` is the number of other corpus documents this file cites; `<-` is the number +that cite it. + +| File | Bytes | `->` | `<-` | Cites (RFC numbers in this corpus) | +|---|---:|---:|---:|---| +| `RFC 2119 - Key Words for Use in RFCs to Indicate Requirement Levels.txt` | 4,723 | 0 | 20 | โ€” | +| `RFC 8174 - Ambiguity of Uppercase vs Lowercase in RFC 2119 Key Words.txt` | 6,071 | 1 | 16 | 2119 | +| `RFC 8126 - Guidelines for Writing an IANA Considerations Section in RFCs.txt` | 109,907 | 1 | 5 | 2119 | +| `RFC 6066 - TLS Extensions Extension Definitions.txt` | 55,079 | 1 | 3 | 2119 | +| `RFC 7301 - TLS Application-Layer Protocol Negotiation Extension.txt` | 17,439 | 1 | 7 | 2119 | +| `RFC 8999 - Version-Independent Properties of QUIC.txt` | 17,393 | 4 | 4 | 2119, 8174, 9000, 9001 | +| `RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt` | 403,442 | 7 | 14 | 2119, 7301, 8126, 8174, 8999, 9001, 9002 | +| `RFC 9001 - Using TLS to Secure QUIC.txt` | 126,175 | 5 | 8 | 2119, 7301, 8174, 9000, 9002 | +| `RFC 9002 - QUIC Loss Detection and Congestion Control.txt` | 89,071 | 4 | 7 | 2119, 8174, 9000, 9001 | +| `RFC 9221 - An Unreliable Datagram Extension to QUIC.txt` | 18,624 | 5 | 3 | 2119, 8174, 9000, 9001, 9002 | +| `RFC 9369 - QUIC Version 2.txt` | 26,887 | 9 | 0 | 2119, 7301, 8174, 8999, 9000, 9001, 9002, 9114, 9250 | +| `RFC 9308 - Applicability of the QUIC Transport Protocol.txt` | 60,645 | 8 | 1 | 7301, 8999, 9000, 9001, 9114, 9218, 9221, 9312 | +| `RFC 9312 - Manageability of the QUIC Transport Protocol.txt` | 80,543 | 9 | 1 | 6066, 7301, 8999, 9000, 9001, 9002, 9114, 9250, 9308 | +| `RFC 9114 - HTTP3.txt` | 155,206 | 7 | 9 | 2119, 6066, 7301, 8126, 8174, 9000, 9204 | +| `RFC 9204 - QPACK Field Compression for HTTP3.txt` | 99,258 | 4 | 1 | 2119, 8174, 9000, 9114 | +| `RFC 9218 - Extensible Prioritization Scheme for HTTP.txt` | 53,974 | 6 | 2 | 2119, 8126, 8174, 9000, 9002, 9114 | +| `RFC 9220 - Bootstrapping WebSockets with HTTP3.txt` | 6,619 | 4 | 2 | 2119, 8174, 8441, 9114 | +| `RFC 9297 - HTTP Datagrams and the Capsule Protocol.txt` | 31,835 | 9 | 1 | 2119, 8126, 8174, 8441, 9000, 9114, 9218, 9220, 9221 | +| `RFC 9298 - Proxying UDP in HTTP.txt` | 37,023 | 8 | 0 | 2119, 8174, 8441, 9000, 9114, 9220, 9221, 9297 | +| `RFC 9412 - The ORIGIN Extension in HTTP3.txt` | 6,879 | 5 | 0 | 2119, 8174, 8336, 9000, 9114 | +| `RFC 8336 - The ORIGIN HTTP2 Frame.txt` | 22,168 | 3 | 1 | 2119, 6066, 8174 | +| `RFC 8441 - Bootstrapping WebSockets with HTTP2.txt` | 16,639 | 2 | 3 | 2119, 8174 | +| `RFC 9250 - DNS over Dedicated QUIC Connections.txt` | 65,667 | 7 | 2 | 2119, 7301, 8126, 8174, 9000, 9001, 9002 | + +Every file's URL is `https://www.rfc-editor.org/rfc/rfc<NNNN>.txt` for the RFC +number in its filename; `download.py` is the authoritative list. + +### Why each document is in the cluster + +| RFC | Why it is here | +|---|---| +| 2119, 8174 | BCP 14. Cited by 20 and 16 of the other 22 files respectively โ€” the corpus's shared boilerplate, and a deliberate hard negative: the phrase "MUST NOT" appears in all 23 documents. | +| 8126 | Defines the registration policies (`Specification Required`, `Expert Review`, `Standards Action`) that the QUIC and HTTP/3 IANA sections *name* without restating. Pure cross-reference material. | +| 6066, 7301 | The two TLS extensions the family builds on: SNI/`server_name` and ALPN. 9001, 9114 and 9250 name their constructs; only these two define them. | +| 8999 | The QUIC invariants. Its 0-255-byte connection-ID range deliberately contradicts 9000's version-1 20-byte cap, which makes version-qualified questions discriminative. | +| 9000 | The hub: cited by 14 of the other 22 files. Also the largest document at 403 kB. | +| 9001 | QUIC's TLS binding. Holds the key-derivation constants 9369 replaces and the Retry Integrity Tag 9000 defers to. | +| 9002 | QUIC loss detection. Two-way deferral with 9000 (PTO โ†” anti-amplification). | +| 9221 | The DATAGRAM extension; the transport layer under 9297's HTTP Datagrams. | +| 9369 | QUIC v2. Defined almost entirely as a diff against 9000/9001/8999 โ€” the densest out-degree in the corpus (9). | +| 9308, 9312 | Applicability and manageability. Cite nearly everything and define nothing, so they are the corpus's "pointer-heavy" documents. | +| 9114 | HTTP/3. Second hub, cited by 9 others. | +| 9204 | QPACK. Its two settings ride in 9114's SETTINGS frame. | +| 9218 | Extensible priorities, in both the HTTP/2 and HTTP/3 spellings. | +| 9220, 8441 | WebSockets over HTTP/3 and over HTTP/2. 9220 is 6.6 kB and reuses 8441's `:protocol` pseudo-header wholesale โ€” the cleanest thin-document/definition pair in the corpus. | +| 9412, 8336 | The ORIGIN extension for HTTP/3 and the ORIGIN frame for HTTP/2. Same pattern as the pair above. | +| 9297, 9298 | HTTP Datagrams / Capsule Protocol and Proxying UDP in HTTP. 9298 defers its data-stream format to 9297. | +| 9250 | DNS over QUIC โ€” a QUIC application protocol that is *not* HTTP, so the corpus is not one protocol stack repeated. | + +## Budget, and what was excluded + +The budget was ~1.5 MB of text so indexing stays tractable. Six documents that +belong to this family on the merits were left out because of it, and their +absence is a real limitation of the corpus, not a neutral choice: + +| Excluded | Bytes | Consequence | +|---|---:|---| +| RFC 9110 (HTTP Semantics) | 502,941 | The most-cited document in the HTTP/3 sub-cluster. 9114, 9111, 9112, 9204 and 9218 all defer core semantics to it; those deferrals are now dangling. | +| RFC 8446 (TLS 1.3) | 337,736 | RFC 9001's principal normative dependency. Cross-reference rows about QUIC packet protection therefore anchor on 7301/6066 instead of on the TLS 1.3 handshake itself. | +| RFC 9113 (HTTP/2) | 191,811 | 8336, 8441 and 9218's HTTP/2 halves refer to it. | +| RFC 6455 (WebSocket) | 162,067 | Referenced by 8441 and 9220. | +| RFC 7541 (HPACK) | 117,827 | QPACK's predecessor, cited by 9204. | +| RFC 9111 / 9112 (HTTP caching, HTTP/1.1) | 84,477 / 109,913 | Same family, dropped for budget. | + +RFC 5234 (ABNF) was in an earlier draft and removed: `download.py --check` +showed it at **degree 0** โ€” no other document in the selected set cites it, +because the family's ABNF references all go through RFC 9110/9112, which are +excluded. RFC 8126 replaced it. + +## Naming + +Files are named `RFC <number> - <Title>.txt`. This is deliberate and it is part +of what is being measured. `rag_system/indexing/crossref.py` resolves a document +mention by searching the corpus text for the document's *whole normalised +filename*, so this scheme resolves a reference only if some document literally +contains the string "RFC 9000 QUIC A UDP Based Multiplexed and Secure +Transport". None does. Measured on the built index: **731 references extracted, +0 resolved.** The alternative schemes were measured too, on the same chunks โ€” +`RFC 9000.txt` resolves 91 mentions across 18 documents, and +`<Title> - RFC 9000.txt` (the word order the RFC references section actually +uses) resolves 53 across 12. (Those figures are the post-fix ones, over the full +683-chunk index; on the pre-fix half-indexed corpus they were 731 / 0, 91 and +35 respectively โ€” the resolution rate was 0 either way.) + +The corpus keeps the plausible-human naming rather than the naming that scores +best, because the finding is the point: see the shakedown report for the full +breakdown. Post-fix, on the fully indexed corpus, the number is **0 of 1403**. + +## What this corpus found on its first run + +Indexing these 23 files with product defaults originally produced a LanceDB +table holding only **52% of the corpus's whitespace-normalised characters**: +every document over ~10,000 markdown tokens retained 45-57%, every document +under it retained ~100%. The cause was a segment-dropping bug in +`MarkdownRecursiveChunker._split_text`, which only fires on documents large +enough to need splitting โ€” which is why no earlier corpus here exposed it (only +`Documentation/design_rationale.md` crosses the threshold, and it was at 49% +retained). **10 of the 24 gold rows failed gate 2** as a direct result. + +That bug was fixed in `rag_system/ingestion/chunking.py` on 2026-08-13. Post-fix, +measured on the same corpus: character retention **1.02** (the >1 is the +chunker's one-sentence overlap), 683 chunks instead of 387, and **gate 2 passes +24/24**. Both states are recorded in the shakedown report. + +The other finding is **not** fixed and is a property of the extractor rather +than of this corpus: index-time cross-reference extraction resolves **0 of 1403** +references here (see *Naming* above). Treat that as the corpus's standing +purpose โ€” it is the only corpus in `eval/` whose reference conventions the +project did not author. diff --git a/eval/corpora/rfc/RFC 2119 - Key Words for Use in RFCs to Indicate Requirement Levels.txt b/eval/corpora/rfc/RFC 2119 - Key Words for Use in RFCs to Indicate Requirement Levels.txt new file mode 100644 index 00000000..e31fae47 --- /dev/null +++ b/eval/corpora/rfc/RFC 2119 - Key Words for Use in RFCs to Indicate Requirement Levels.txt @@ -0,0 +1,171 @@ + + + + + + +Network Working Group S. Bradner +Request for Comments: 2119 Harvard University +BCP: 14 March 1997 +Category: Best Current Practice + + + Key words for use in RFCs to Indicate Requirement Levels + +Status of this Memo + + This document specifies an Internet Best Current Practices for the + Internet Community, and requests discussion and suggestions for + improvements. Distribution of this memo is unlimited. + +Abstract + + In many standards track documents several words are used to signify + the requirements in the specification. These words are often + capitalized. This document defines these words as they should be + interpreted in IETF documents. Authors who follow these guidelines + should incorporate this phrase near the beginning of their document: + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL + NOT", "SHOULD", "SHOULD NOT", "RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + RFC 2119. + + Note that the force of these words is modified by the requirement + level of the document in which they are used. + +1. MUST This word, or the terms "REQUIRED" or "SHALL", mean that the + definition is an absolute requirement of the specification. + +2. MUST NOT This phrase, or the phrase "SHALL NOT", mean that the + definition is an absolute prohibition of the specification. + +3. SHOULD This word, or the adjective "RECOMMENDED", mean that there + may exist valid reasons in particular circumstances to ignore a + particular item, but the full implications must be understood and + carefully weighed before choosing a different course. + +4. SHOULD NOT This phrase, or the phrase "NOT RECOMMENDED" mean that + there may exist valid reasons in particular circumstances when the + particular behavior is acceptable or even useful, but the full + implications should be understood and the case carefully weighed + before implementing any behavior described with this label. + + + + + +Bradner Best Current Practice [Page 1] + +RFC 2119 RFC Key Words March 1997 + + +5. MAY This word, or the adjective "OPTIONAL", mean that an item is + truly optional. One vendor may choose to include the item because a + particular marketplace requires it or because the vendor feels that + it enhances the product while another vendor may omit the same item. + An implementation which does not include a particular option MUST be + prepared to interoperate with another implementation which does + include the option, though perhaps with reduced functionality. In the + same vein an implementation which does include a particular option + MUST be prepared to interoperate with another implementation which + does not include the option (except, of course, for the feature the + option provides.) + +6. Guidance in the use of these Imperatives + + Imperatives of the type defined in this memo must be used with care + and sparingly. In particular, they MUST only be used where it is + actually required for interoperation or to limit behavior which has + potential for causing harm (e.g., limiting retransmisssions) For + example, they must not be used to try to impose a particular method + on implementors where the method is not required for + interoperability. + +7. Security Considerations + + These terms are frequently used to specify behavior with security + implications. The effects on security of not implementing a MUST or + SHOULD, or doing something the specification says MUST NOT or SHOULD + NOT be done may be very subtle. Document authors should take the time + to elaborate the security implications of not following + recommendations or requirements as most implementors will not have + had the benefit of the experience and discussion that produced the + specification. + +8. Acknowledgments + + The definitions of these terms are an amalgam of definitions taken + from a number of RFCs. In addition, suggestions have been + incorporated from a number of people including Robert Ullmann, Thomas + Narten, Neal McBurnett, and Robert Elz. + + + + + + + + + + + + +Bradner Best Current Practice [Page 2] + +RFC 2119 RFC Key Words March 1997 + + +9. Author's Address + + Scott Bradner + Harvard University + 1350 Mass. Ave. + Cambridge, MA 02138 + + phone - +1 617 495 3864 + + email - sob@harvard.edu + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Bradner Best Current Practice [Page 3] + diff --git a/eval/corpora/rfc/RFC 6066 - TLS Extensions Extension Definitions.txt b/eval/corpora/rfc/RFC 6066 - TLS Extensions Extension Definitions.txt new file mode 100644 index 00000000..c1b777ca --- /dev/null +++ b/eval/corpora/rfc/RFC 6066 - TLS Extensions Extension Definitions.txt @@ -0,0 +1,1403 @@ + + + + + + +Internet Engineering Task Force (IETF) D. Eastlake 3rd +Request for Comments: 6066 Huawei +Obsoletes: 4366 January 2011 +Category: Standards Track +ISSN: 2070-1721 + + + Transport Layer Security (TLS) Extensions: Extension Definitions + +Abstract + + This document provides specifications for existing TLS extensions. + It is a companion document for RFC 5246, "The Transport Layer + Security (TLS) Protocol Version 1.2". The extensions specified are + server_name, max_fragment_length, client_certificate_url, + trusted_ca_keys, truncated_hmac, and status_request. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 5741. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + http://www.rfc-editor.org/info/rfc6066. + +Copyright Notice + + Copyright (c) 2011 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (http://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + + + + + + +Eastlake Standards Track [Page 1] + +RFC 6066 TLS Extension Definitions January 2011 + + + This document may contain material from IETF Documents or IETF + Contributions published or made publicly available before November + 10, 2008. The person(s) controlling the copyright in some of this + material may not have granted the IETF Trust the right to allow + modifications of such material outside the IETF Standards Process. + Without obtaining an adequate license from the person(s) controlling + the copyright in such materials, this document may not be modified + outside the IETF Standards Process, and derivative works of it may + not be created outside the IETF Standards Process, except to format + it for publication as an RFC or to translate it into languages other + than English. + +Table of Contents + + 1. Introduction ....................................................3 + 1.1. Specific Extensions Covered ................................3 + 1.2. Conventions Used in This Document ..........................5 + 2. Extensions to the Handshake Protocol ............................5 + 3. Server Name Indication ..........................................6 + 4. Maximum Fragment Length Negotiation .............................8 + 5. Client Certificate URLs .........................................9 + 6. Trusted CA Indication ..........................................12 + 7. Truncated HMAC .................................................13 + 8. Certificate Status Request .....................................14 + 9. Error Alerts ...................................................16 + 10. IANA Considerations ...........................................17 + 10.1. pkipath MIME Type Registration ...........................17 + 10.2. Reference for TLS Alerts, TLS HandshakeTypes, and + ExtensionTypes ...........................................19 + 11. Security Considerations .......................................19 + 11.1. Security Considerations for server_name ..................19 + 11.2. Security Considerations for max_fragment_length ..........20 + 11.3. Security Considerations for client_certificate_url .......20 + 11.4. Security Considerations for trusted_ca_keys ..............21 + 11.5. Security Considerations for truncated_hmac ...............21 + 11.6. Security Considerations for status_request ...............22 + 12. Normative References ..........................................22 + 13. Informative References ........................................23 + Appendix A. Changes from RFC 4366 .................................24 + Appendix B. Acknowledgements ......................................25 + + + + + + + + + + + +Eastlake Standards Track [Page 2] + +RFC 6066 TLS Extension Definitions January 2011 + + +1. Introduction + + The Transport Layer Security (TLS) Protocol Version 1.2 is specified + in [RFC5246]. That specification includes the framework for + extensions to TLS, considerations in designing such extensions (see + Section 7.4.1.4 of [RFC5246]), and IANA Considerations for the + allocation of new extension code points; however, it does not specify + any particular extensions other than Signature Algorithms (see + Section 7.4.1.4.1 of [RFC5246]). + + This document provides the specifications for existing TLS + extensions. It is, for the most part, the adaptation and editing of + material from RFC 4366, which covered TLS extensions for TLS 1.0 (RFC + 2246) and TLS 1.1 (RFC 4346). + +1.1. Specific Extensions Covered + + The extensions described here focus on extending the functionality + provided by the TLS protocol message formats. Other issues, such as + the addition of new cipher suites, are deferred. + + The extension types defined in this document are: + + enum { + server_name(0), max_fragment_length(1), + client_certificate_url(2), trusted_ca_keys(3), + truncated_hmac(4), status_request(5), (65535) + } ExtensionType; + + Specifically, the extensions described in this document: + + - Allow TLS clients to provide to the TLS server the name of the + server they are contacting. This functionality is desirable in + order to facilitate secure connections to servers that host + multiple 'virtual' servers at a single underlying network address. + + - Allow TLS clients and servers to negotiate the maximum fragment + length to be sent. This functionality is desirable as a result of + memory constraints among some clients, and bandwidth constraints + among some access networks. + + - Allow TLS clients and servers to negotiate the use of client + certificate URLs. This functionality is desirable in order to + conserve memory on constrained clients. + + + + + + + +Eastlake Standards Track [Page 3] + +RFC 6066 TLS Extension Definitions January 2011 + + + - Allow TLS clients to indicate to TLS servers which certification + authority (CA) root keys they possess. This functionality is + desirable in order to prevent multiple handshake failures + involving TLS clients that are only able to store a small number + of CA root keys due to memory limitations. + + - Allow TLS clients and servers to negotiate the use of truncated + Message Authentication Codes (MACs). This functionality is + desirable in order to conserve bandwidth in constrained access + networks. + + - Allow TLS clients and servers to negotiate that the server sends + the client certificate status information (e.g., an Online + Certificate Status Protocol (OCSP) [RFC2560] response) during a + TLS handshake. This functionality is desirable in order to avoid + sending a Certificate Revocation List (CRL) over a constrained + access network and therefore saving bandwidth. + + TLS clients and servers may use the extensions described in this + document. The extensions are designed to be backwards compatible, + meaning that TLS clients that support the extensions can talk to TLS + servers that do not support the extensions, and vice versa. + + Note that any messages associated with these extensions that are sent + during the TLS handshake MUST be included in the hash calculations + involved in "Finished" messages. + + Note also that all the extensions defined in this document are + relevant only when a session is initiated. A client that requests + session resumption does not in general know whether the server will + accept this request, and therefore it SHOULD send the same extensions + as it would send if it were not attempting resumption. When a client + includes one or more of the defined extension types in an extended + client hello while requesting session resumption: + + - The server name indication extension MAY be used by the server + when deciding whether or not to resume a session as described in + Section 3. + + - If the resumption request is denied, the use of the extensions is + negotiated as normal. + + - If, on the other hand, the older session is resumed, then the + server MUST ignore the extensions and send a server hello + containing none of the extension types. In this case, the + functionality of these extensions negotiated during the original + session initiation is applied to the resumed session. + + + + +Eastlake Standards Track [Page 4] + +RFC 6066 TLS Extension Definitions January 2011 + + +1.2. Conventions Used in This Document + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + [RFC2119]. + +2. Extensions to the Handshake Protocol + + This document specifies the use of two new handshake messages, + "CertificateURL" and "CertificateStatus". These messages are + described in Sections 5 and 8, respectively. The new handshake + message structure therefore becomes: + + enum { + hello_request(0), client_hello(1), server_hello(2), + certificate(11), server_key_exchange (12), + certificate_request(13), server_hello_done(14), + certificate_verify(15), client_key_exchange(16), + finished(20), certificate_url(21), certificate_status(22), + (255) + } HandshakeType; + + struct { + HandshakeType msg_type; /* handshake type */ + uint24 length; /* bytes in message */ + select (HandshakeType) { + case hello_request: HelloRequest; + case client_hello: ClientHello; + case server_hello: ServerHello; + case certificate: Certificate; + case server_key_exchange: ServerKeyExchange; + case certificate_request: CertificateRequest; + case server_hello_done: ServerHelloDone; + case certificate_verify: CertificateVerify; + case client_key_exchange: ClientKeyExchange; + case finished: Finished; + case certificate_url: CertificateURL; + case certificate_status: CertificateStatus; + } body; + } Handshake; + + + + + + + + + + +Eastlake Standards Track [Page 5] + +RFC 6066 TLS Extension Definitions January 2011 + + +3. Server Name Indication + + TLS does not provide a mechanism for a client to tell a server the + name of the server it is contacting. It may be desirable for clients + to provide this information to facilitate secure connections to + servers that host multiple 'virtual' servers at a single underlying + network address. + + In order to provide any of the server names, clients MAY include an + extension of type "server_name" in the (extended) client hello. The + "extension_data" field of this extension SHALL contain + "ServerNameList" where: + + struct { + NameType name_type; + select (name_type) { + case host_name: HostName; + } name; + } ServerName; + + enum { + host_name(0), (255) + } NameType; + + opaque HostName<1..2^16-1>; + + struct { + ServerName server_name_list<1..2^16-1> + } ServerNameList; + + The ServerNameList MUST NOT contain more than one name of the same + name_type. If the server understood the ClientHello extension but + does not recognize the server name, the server SHOULD take one of two + actions: either abort the handshake by sending a fatal-level + unrecognized_name(112) alert or continue the handshake. It is NOT + RECOMMENDED to send a warning-level unrecognized_name(112) alert, + because the client's behavior in response to warning-level alerts is + unpredictable. If there is a mismatch between the server name used + by the client application and the server name of the credential + chosen by the server, this mismatch will become apparent when the + client application performs the server endpoint identification, at + which point the client application will have to decide whether to + proceed with the communication. TLS implementations are encouraged + to make information available to application callers about warning- + level alerts that were received or sent during a TLS handshake. Such + information can be useful for diagnostic purposes. + + + + + +Eastlake Standards Track [Page 6] + +RFC 6066 TLS Extension Definitions January 2011 + + + Note: Earlier versions of this specification permitted multiple + names of the same name_type. In practice, current client + implementations only send one name, and the client cannot + necessarily find out which name the server selected. Multiple + names of the same name_type are therefore now prohibited. + + Currently, the only server names supported are DNS hostnames; + however, this does not imply any dependency of TLS on DNS, and other + name types may be added in the future (by an RFC that updates this + document). The data structure associated with the host_name NameType + is a variable-length vector that begins with a 16-bit length. For + backward compatibility, all future data structures associated with + new NameTypes MUST begin with a 16-bit length field. TLS MAY treat + provided server names as opaque data and pass the names and types to + the application. + + "HostName" contains the fully qualified DNS hostname of the server, + as understood by the client. The hostname is represented as a byte + string using ASCII encoding without a trailing dot. This allows the + support of internationalized domain names through the use of A-labels + defined in [RFC5890]. DNS hostnames are case-insensitive. The + algorithm to compare hostnames is described in [RFC5890], Section + 2.3.2.4. + + Literal IPv4 and IPv6 addresses are not permitted in "HostName". + + It is RECOMMENDED that clients include an extension of type + "server_name" in the client hello whenever they locate a server by a + supported name type. + + A server that receives a client hello containing the "server_name" + extension MAY use the information contained in the extension to guide + its selection of an appropriate certificate to return to the client, + and/or other aspects of security policy. In this event, the server + SHALL include an extension of type "server_name" in the (extended) + server hello. The "extension_data" field of this extension SHALL be + empty. + + When the server is deciding whether or not to accept a request to + resume a session, the contents of a server_name extension MAY be used + in the lookup of the session in the session cache. The client SHOULD + include the same server_name extension in the session resumption + request as it did in the full handshake that established the session. + A server that implements this extension MUST NOT accept the request + to resume the session if the server_name extension contains a + different name. Instead, it proceeds with a full handshake to + establish a new session. When resuming a session, the server MUST + NOT include a server_name extension in the server hello. + + + +Eastlake Standards Track [Page 7] + +RFC 6066 TLS Extension Definitions January 2011 + + + If an application negotiates a server name using an application + protocol and then upgrades to TLS, and if a server_name extension is + sent, then the extension SHOULD contain the same name that was + negotiated in the application protocol. If the server_name is + established in the TLS session handshake, the client SHOULD NOT + attempt to request a different server name at the application layer. + +4. Maximum Fragment Length Negotiation + + Without this extension, TLS specifies a fixed maximum plaintext + fragment length of 2^14 bytes. It may be desirable for constrained + clients to negotiate a smaller maximum fragment length due to memory + limitations or bandwidth limitations. + + In order to negotiate smaller maximum fragment lengths, clients MAY + include an extension of type "max_fragment_length" in the (extended) + client hello. The "extension_data" field of this extension SHALL + contain: + + enum{ + 2^9(1), 2^10(2), 2^11(3), 2^12(4), (255) + } MaxFragmentLength; + + whose value is the desired maximum fragment length. The allowed + values for this field are: 2^9, 2^10, 2^11, and 2^12. + + Servers that receive an extended client hello containing a + "max_fragment_length" extension MAY accept the requested maximum + fragment length by including an extension of type + "max_fragment_length" in the (extended) server hello. The + "extension_data" field of this extension SHALL contain a + "MaxFragmentLength" whose value is the same as the requested maximum + fragment length. + + If a server receives a maximum fragment length negotiation request + for a value other than the allowed values, it MUST abort the + handshake with an "illegal_parameter" alert. Similarly, if a client + receives a maximum fragment length negotiation response that differs + from the length it requested, it MUST also abort the handshake with + an "illegal_parameter" alert. + + Once a maximum fragment length other than 2^14 has been successfully + negotiated, the client and server MUST immediately begin fragmenting + messages (including handshake messages) to ensure that no fragment + larger than the negotiated length is sent. Note that TLS already + requires clients and servers to support fragmentation of handshake + messages. + + + + +Eastlake Standards Track [Page 8] + +RFC 6066 TLS Extension Definitions January 2011 + + + The negotiated length applies for the duration of the session + including session resumptions. + + The negotiated length limits the input that the record layer may + process without fragmentation (that is, the maximum value of + TLSPlaintext.length; see [RFC5246], Section 6.2.1). Note that the + output of the record layer may be larger. For example, if the + negotiated length is 2^9=512, then, when using currently defined + cipher suites (those defined in [RFC5246] and [RFC2712]) and null + compression, the record-layer output can be at most 805 bytes: 5 + bytes of headers, 512 bytes of application data, 256 bytes of + padding, and 32 bytes of MAC. This means that in this event a TLS + record-layer peer receiving a TLS record-layer message larger than + 805 bytes MUST discard the message and send a "record_overflow" + alert, without decrypting the message. When this extension is used + with Datagram Transport Layer Security (DTLS), implementations SHOULD + NOT generate record_overflow alerts unless the packet passes message + authentication. + +5. Client Certificate URLs + + Without this extension, TLS specifies that when client authentication + is performed, client certificates are sent by clients to servers + during the TLS handshake. It may be desirable for constrained + clients to send certificate URLs in place of certificates, so that + they do not need to store their certificates and can therefore save + memory. + + In order to negotiate sending certificate URLs to a server, clients + MAY include an extension of type "client_certificate_url" in the + (extended) client hello. The "extension_data" field of this + extension SHALL be empty. + + (Note that it is necessary to negotiate the use of client certificate + URLs in order to avoid "breaking" existing TLS servers.) + + Servers that receive an extended client hello containing a + "client_certificate_url" extension MAY indicate that they are willing + to accept certificate URLs by including an extension of type + "client_certificate_url" in the (extended) server hello. The + "extension_data" field of this extension SHALL be empty. + + After negotiation of the use of client certificate URLs has been + successfully completed (by exchanging hellos including + "client_certificate_url" extensions), clients MAY send a + "CertificateURL" message in place of a "Certificate" message as + follows (see also Section 2): + + + + +Eastlake Standards Track [Page 9] + +RFC 6066 TLS Extension Definitions January 2011 + + + enum { + individual_certs(0), pkipath(1), (255) + } CertChainType; + + struct { + CertChainType type; + URLAndHash url_and_hash_list<1..2^16-1>; + } CertificateURL; + + struct { + opaque url<1..2^16-1>; + unint8 padding; + opaque SHA1Hash[20]; + } URLAndHash; + + Here, "url_and_hash_list" contains a sequence of URLs and hashes. + Each "url" MUST be an absolute URI reference according to [RFC3986] + that can be immediately used to fetch the certificate(s). + + When X.509 certificates are used, there are two possibilities: + + - If CertificateURL.type is "individual_certs", each URL refers to a + single DER-encoded X.509v3 certificate, with the URL for the + client's certificate first. + + - If CertificateURL.type is "pkipath", the list contains a single + URL referring to a DER-encoded certificate chain, using the type + PkiPath described in Section 10.1. + + When any other certificate format is used, the specification that + describes use of that format in TLS should define the encoding format + of certificates or certificate chains, and any constraint on their + ordering. + + The "padding" byte MUST be 0x01. It is present to make the structure + backwards compatible. + + The hash corresponding to each URL is the SHA-1 hash of the + certificate or certificate chain (in the case of X.509 certificates, + the DER-encoded certificate or the DER-encoded PkiPath). + + Note that when a list of URLs for X.509 certificates is used, the + ordering of URLs is the same as that used in the TLS Certificate + message (see [RFC5246], Section 7.4.2), but opposite to the order in + which certificates are encoded in PkiPath. In either case, the self- + signed root certificate MAY be omitted from the chain, under the + assumption that the server must already possess it in order to + validate it. + + + +Eastlake Standards Track [Page 10] + +RFC 6066 TLS Extension Definitions January 2011 + + + Servers receiving "CertificateURL" SHALL attempt to retrieve the + client's certificate chain from the URLs and then process the + certificate chain as usual. A cached copy of the content of any URL + in the chain MAY be used, provided that the SHA-1 hash matches the + hash of the cached copy. + + Servers that support this extension MUST support the 'http' URI + scheme for certificate URLs and MAY support other schemes. Use of + other schemes than 'http', 'https', or 'ftp' may create unexpected + problems. + + If the protocol used is HTTP, then the HTTP server can be configured + to use the Cache-Control and Expires directives described in + [RFC2616] to specify whether and for how long certificates or + certificate chains should be cached. + + The TLS server MUST NOT follow HTTP redirects when retrieving the + certificates or certificate chain. The URLs used in this extension + MUST NOT be chosen to depend on such redirects. + + If the protocol used to retrieve certificates or certificate chains + returns a MIME-formatted response (as HTTP does), then the following + MIME Content-Types SHALL be used: when a single X.509v3 certificate + is returned, the Content-Type is "application/pkix-cert" [RFC2585], + and when a chain of X.509v3 certificates is returned, the Content- + Type is "application/pkix-pkipath" (Section 10.1). + + The server MUST check that the SHA-1 hash of the contents of the + object retrieved from that URL (after decoding any MIME Content- + Transfer-Encoding) matches the given hash. If any retrieved object + does not have the correct SHA-1 hash, the server MUST abort the + handshake with a bad_certificate_hash_value(114) alert. This alert + is always fatal. + + Clients may choose to send either "Certificate" or "CertificateURL" + after successfully negotiating the option to send certificate URLs. + The option to send a certificate is included to provide flexibility + to clients possessing multiple certificates. + + If a server is unable to obtain certificates in a given + CertificateURL, it MUST send a fatal certificate_unobtainable(111) + alert if it requires the certificates to complete the handshake. If + the server does not require the certificates, then the server + continues the handshake. The server MAY send a warning-level alert + in this case. Clients receiving such an alert SHOULD log the alert + and continue with the handshake if possible. + + + + + +Eastlake Standards Track [Page 11] + +RFC 6066 TLS Extension Definitions January 2011 + + +6. Trusted CA Indication + + Constrained clients that, due to memory limitations, possess only a + small number of CA root keys may wish to indicate to servers which + root keys they possess, in order to avoid repeated handshake + failures. + + In order to indicate which CA root keys they possess, clients MAY + include an extension of type "trusted_ca_keys" in the (extended) + client hello. The "extension_data" field of this extension SHALL + contain "TrustedAuthorities" where: + + struct { + TrustedAuthority trusted_authorities_list<0..2^16-1>; + } TrustedAuthorities; + + struct { + IdentifierType identifier_type; + select (identifier_type) { + case pre_agreed: struct {}; + case key_sha1_hash: SHA1Hash; + case x509_name: DistinguishedName; + case cert_sha1_hash: SHA1Hash; + } identifier; + } TrustedAuthority; + + enum { + pre_agreed(0), key_sha1_hash(1), x509_name(2), + cert_sha1_hash(3), (255) + } IdentifierType; + + opaque DistinguishedName<1..2^16-1>; + + Here, "TrustedAuthorities" provides a list of CA root key identifiers + that the client possesses. Each CA root key is identified via + either: + + - "pre_agreed": no CA root key identity supplied. + + - "key_sha1_hash": contains the SHA-1 hash of the CA root key. For + Digital Signature Algorithm (DSA) and Elliptic Curve Digital + Signature Algorithm (ECDSA) keys, this is the hash of the + "subjectPublicKey" value. For RSA keys, the hash is of the big- + endian byte string representation of the modulus without any + initial zero-valued bytes. (This copies the key hash formats + deployed in other environments.) + + + + + +Eastlake Standards Track [Page 12] + +RFC 6066 TLS Extension Definitions January 2011 + + + - "x509_name": contains the DER-encoded X.509 DistinguishedName of + the CA. + + - "cert_sha1_hash": contains the SHA-1 hash of a DER-encoded + Certificate containing the CA root key. + + Note that clients may include none, some, or all of the CA root keys + they possess in this extension. + + Note also that it is possible that a key hash or a Distinguished Name + alone may not uniquely identify a certificate issuer (for example, if + a particular CA has multiple key pairs). However, here we assume + this is the case following the use of Distinguished Names to identify + certificate issuers in TLS. + + The option to include no CA root keys is included to allow the client + to indicate possession of some pre-defined set of CA root keys. + + Servers that receive a client hello containing the "trusted_ca_keys" + extension MAY use the information contained in the extension to guide + their selection of an appropriate certificate chain to return to the + client. In this event, the server SHALL include an extension of type + "trusted_ca_keys" in the (extended) server hello. The + "extension_data" field of this extension SHALL be empty. + +7. Truncated HMAC + + Currently defined TLS cipher suites use the MAC construction HMAC + [RFC2104] to authenticate record-layer communications. In TLS, the + entire output of the hash function is used as the MAC tag. However, + it may be desirable in constrained environments to save bandwidth by + truncating the output of the hash function to 80 bits when forming + MAC tags. + + In order to negotiate the use of 80-bit truncated HMAC, clients MAY + include an extension of type "truncated_hmac" in the extended client + hello. The "extension_data" field of this extension SHALL be empty. + + Servers that receive an extended hello containing a "truncated_hmac" + extension MAY agree to use a truncated HMAC by including an extension + of type "truncated_hmac", with empty "extension_data", in the + extended server hello. + + Note that if new cipher suites are added that do not use HMAC, and + the session negotiates one of these cipher suites, this extension + will have no effect. It is strongly recommended that any new cipher + suites using other MACs consider the MAC size an integral part of the + + + + +Eastlake Standards Track [Page 13] + +RFC 6066 TLS Extension Definitions January 2011 + + + cipher suite definition, taking into account both security and + bandwidth considerations. + + If HMAC truncation has been successfully negotiated during a TLS + handshake, and the negotiated cipher suite uses HMAC, both the client + and the server pass this fact to the TLS record layer along with the + other negotiated security parameters. Subsequently during the + session, clients and servers MUST use truncated HMACs, calculated as + specified in [RFC2104]. That is, SecurityParameters.mac_length is 10 + bytes, and only the first 10 bytes of the HMAC output are transmitted + and checked. Note that this extension does not affect the + calculation of the pseudo-random function (PRF) as part of + handshaking or key derivation. + + The negotiated HMAC truncation size applies for the duration of the + session including session resumptions. + +8. Certificate Status Request + + Constrained clients may wish to use a certificate-status protocol + such as OCSP [RFC2560] to check the validity of server certificates, + in order to avoid transmission of CRLs and therefore save bandwidth + on constrained networks. This extension allows for such information + to be sent in the TLS handshake, saving roundtrips and resources. + + In order to indicate their desire to receive certificate status + information, clients MAY include an extension of type + "status_request" in the (extended) client hello. The + "extension_data" field of this extension SHALL contain + "CertificateStatusRequest" where: + + struct { + CertificateStatusType status_type; + select (status_type) { + case ocsp: OCSPStatusRequest; + } request; + } CertificateStatusRequest; + + enum { ocsp(1), (255) } CertificateStatusType; + + struct { + ResponderID responder_id_list<0..2^16-1>; + Extensions request_extensions; + } OCSPStatusRequest; + + opaque ResponderID<1..2^16-1>; + opaque Extensions<0..2^16-1>; + + + + +Eastlake Standards Track [Page 14] + +RFC 6066 TLS Extension Definitions January 2011 + + + In the OCSPStatusRequest, the "ResponderIDs" provides a list of OCSP + responders that the client trusts. A zero-length "responder_id_list" + sequence has the special meaning that the responders are implicitly + known to the server, e.g., by prior arrangement. "Extensions" is a + DER encoding of OCSP request extensions. + + Both "ResponderID" and "Extensions" are DER-encoded ASN.1 types as + defined in [RFC2560]. "Extensions" is imported from [RFC5280]. A + zero-length "request_extensions" value means that there are no + extensions (as opposed to a zero-length ASN.1 SEQUENCE, which is not + valid for the "Extensions" type). + + In the case of the "id-pkix-ocsp-nonce" OCSP extension, [RFC2560] is + unclear about its encoding; for clarification, the nonce MUST be a + DER-encoded OCTET STRING, which is encapsulated as another OCTET + STRING (note that implementations based on an existing OCSP client + will need to be checked for conformance to this requirement). + + Servers that receive a client hello containing the "status_request" + extension MAY return a suitable certificate status response to the + client along with their certificate. If OCSP is requested, they + SHOULD use the information contained in the extension when selecting + an OCSP responder and SHOULD include request_extensions in the OCSP + request. + + Servers return a certificate response along with their certificate by + sending a "CertificateStatus" message immediately after the + "Certificate" message (and before any "ServerKeyExchange" or + "CertificateRequest" messages). If a server returns a + "CertificateStatus" message, then the server MUST have included an + extension of type "status_request" with empty "extension_data" in the + extended server hello. The "CertificateStatus" message is conveyed + using the handshake message type "certificate_status" as follows (see + also Section 2): + + struct { + CertificateStatusType status_type; + select (status_type) { + case ocsp: OCSPResponse; + } response; + } CertificateStatus; + + opaque OCSPResponse<1..2^24-1>; + + An "ocsp_response" contains a complete, DER-encoded OCSP response + (using the ASN.1 type OCSPResponse defined in [RFC2560]). Only one + OCSP response may be sent. + + + + +Eastlake Standards Track [Page 15] + +RFC 6066 TLS Extension Definitions January 2011 + + + Note that a server MAY also choose not to send a "CertificateStatus" + message, even if has received a "status_request" extension in the + client hello message and has sent a "status_request" extension in the + server hello message. + + Note in addition that a server MUST NOT send the "CertificateStatus" + message unless it received a "status_request" extension in the client + hello message and sent a "status_request" extension in the server + hello message. + + Clients requesting an OCSP response and receiving an OCSP response in + a "CertificateStatus" message MUST check the OCSP response and abort + the handshake if the response is not satisfactory with + bad_certificate_status_response(113) alert. This alert is always + fatal. + +9. Error Alerts + + Four new error alerts are defined for use with the TLS extensions + defined in this document. To avoid "breaking" existing clients and + servers, these alerts MUST NOT be sent unless the sending party has + received an extended hello message from the party they are + communicating with. These error alerts are conveyed using the + following syntax. The new alerts are the last four, as indicated by + the comments on the same line as the error alert number. + + enum { + close_notify(0), + unexpected_message(10), + bad_record_mac(20), + decryption_failed(21), + record_overflow(22), + decompression_failure(30), + handshake_failure(40), + /* 41 is not defined, for historical reasons */ + bad_certificate(42), + unsupported_certificate(43), + certificate_revoked(44), + certificate_expired(45), + certificate_unknown(46), + illegal_parameter(47), + unknown_ca(48), + access_denied(49), + decode_error(50), + decrypt_error(51), + export_restriction(60), + protocol_version(70), + insufficient_security(71), + + + +Eastlake Standards Track [Page 16] + +RFC 6066 TLS Extension Definitions January 2011 + + + internal_error(80), + user_canceled(90), + no_renegotiation(100), + unsupported_extension(110), + certificate_unobtainable(111), /* new */ + unrecognized_name(112), /* new */ + bad_certificate_status_response(113), /* new */ + bad_certificate_hash_value(114), /* new */ + (255) + } AlertDescription; + + "certificate_unobtainable" is described in Section 5. + "unrecognized_name" is described in Section 3. + "bad_certificate_status_response" is described in Section 8. + "bad_certificate_hash_value" is described in Section 5. + +10. IANA Considerations + + IANA Considerations for TLS extensions and the creation of a registry + are covered in Section 12 of [RFC5246] except for the registration of + MIME type application/pkix-pkipath, which appears below. + + The IANA TLS extensions and MIME type application/pkix-pkipath + registry entries that reference RFC 4366 have been updated to + reference this document. + +10.1. pkipath MIME Type Registration + + MIME media type name: application + MIME subtype name: pkix-pkipath + Required parameters: none + + Optional parameters: version (default value is "1") + + Encoding considerations: + Binary; this MIME type is a DER encoding of the ASN.1 type + PkiPath, defined as follows: + PkiPath ::= SEQUENCE OF Certificate + PkiPath is used to represent a certification path. Within the + sequence, the order of certificates is such that the subject of + the first certificate is the issuer of the second certificate, + etc. + This is identical to the definition published in [X509-4th-TC1]; + note that it is different from that in [X509-4th]. + + All Certificates MUST conform to [RFC5280]. (This should be + interpreted as a requirement to encode only PKIX-conformant + certificates using this type. It does not necessarily require + + + +Eastlake Standards Track [Page 17] + +RFC 6066 TLS Extension Definitions January 2011 + + + that all certificates that are not strictly PKIX-conformant must + be rejected by relying parties, although the security consequences + of accepting any such certificates should be considered + carefully.) + + DER (as opposed to BER) encoding MUST be used. If this type is + sent over a 7-bit transport, base64 encoding SHOULD be used. + + Security considerations: + The security considerations of [X509-4th] and [RFC5280] (or any + updates to them) apply, as well as those of any protocol that uses + this type (e.g., TLS). + + Note that this type only specifies a certificate chain that can be + assessed for validity according to the relying party's existing + configuration of trusted CAs; it is not intended to be used to + specify any change to that configuration. + + Interoperability considerations: + No specific interoperability problems are known with this type, + but for recommendations relating to X.509 certificates in general, + see [RFC5280]. + + Published specification: This document and [RFC5280]. + + Applications that use this media type: + TLS. It may also be used by other protocols or for general + interchange of PKIX certificate chains. + + Additional information: + Magic number(s): DER-encoded ASN.1 can be easily recognized. + Further parsing is required to distinguish it from other ASN.1 + types. + File extension(s): .pkipath + Macintosh File Type Code(s): not specified + + Person & email address to contact for further information: + Magnus Nystrom <mnystrom@microsoft.com> + + Intended usage: COMMON + + Change controller: IESG <iesg@ietf.org> + + + + + + + + + +Eastlake Standards Track [Page 18] + +RFC 6066 TLS Extension Definitions January 2011 + + +10.2. Reference for TLS Alerts, TLS HandshakeTypes, and ExtensionTypes + + The following values in the TLS Alert Registry have been updated to + reference this document: + + 111 certificate_unobtainable + 112 unrecognized_name + 113 bad_certificate_status_response + 114 bad_certificate_hash_value + + The following values in the TLS HandshakeType Registry have been + updated to reference this document: + + 21 certificate_url + 22 certificate_status + + The following ExtensionType values have been updated to reference + this document: + + 0 server_name + 1 max_fragment_length + 2 client_certificate_url + 3 trusted_ca_keys + 4 truncated_hmac + 5 status_request + +11. Security Considerations + + General security considerations for TLS extensions are covered in + [RFC5246]. Security Considerations for particular extensions + specified in this document are given below. + + In general, implementers should continue to monitor the state of the + art and address any weaknesses identified. + +11.1. Security Considerations for server_name + + If a single server hosts several domains, then clearly it is + necessary for the owners of each domain to ensure that this satisfies + their security needs. Apart from this, server_name does not appear + to introduce significant security issues. + + Since it is possible for a client to present a different server_name + in the application protocol, application server implementations that + rely upon these names being the same MUST check to make sure the + client did not present a different name in the application protocol. + + + + + +Eastlake Standards Track [Page 19] + +RFC 6066 TLS Extension Definitions January 2011 + + + Implementations MUST ensure that a buffer overflow does not occur, + whatever the values of the length fields in server_name. + +11.2. Security Considerations for max_fragment_length + + The maximum fragment length takes effect immediately, including for + handshake messages. However, that does not introduce any security + complications that are not already present in TLS, since TLS requires + implementations to be able to handle fragmented handshake messages. + + Note that, as described in Section 4, once a non-null cipher suite + has been activated, the effective maximum fragment length depends on + the cipher suite and compression method, as well as on the negotiated + max_fragment_length. This must be taken into account when sizing + buffers and checking for buffer overflow. + +11.3. Security Considerations for client_certificate_url + + Support for client_certificate_url involves the server's acting as a + client in another URI-scheme-dependent protocol. The server + therefore becomes subject to many of the same security concerns that + clients of the URI scheme are subject to, with the added concern that + the client can attempt to prompt the server to connect to some + (possibly weird-looking) URL. + + In general, this issue means that an attacker might use the server to + indirectly attack another host that is vulnerable to some security + flaw. It also introduces the possibility of denial-of-service + attacks in which an attacker makes many connections to the server, + each of which results in the server's attempting a connection to the + target of the attack. + + Note that the server may be behind a firewall or otherwise able to + access hosts that would not be directly accessible from the public + Internet. This could exacerbate the potential security and denial- + of-service problems described above, as well as allow the existence + of internal hosts to be confirmed when they would otherwise be + hidden. + + The detailed security concerns involved will depend on the URI + schemes supported by the server. In the case of HTTP, the concerns + are similar to those that apply to a publicly accessible HTTP proxy + server. In the case of HTTPS, loops and deadlocks may be created, + and this should be addressed. In the case of FTP, attacks arise that + are similar to FTP bounce attacks. + + + + + + +Eastlake Standards Track [Page 20] + +RFC 6066 TLS Extension Definitions January 2011 + + + As a result of this issue, it is RECOMMENDED that the + client_certificate_url extension should have to be specifically + enabled by a server administrator, rather than be enabled by default. + It is also RECOMMENDED that URI schemes be enabled by the + administrator individually, and only a minimal set of schemes be + enabled. Unusual protocols that offer limited security or whose + security is not well understood SHOULD be avoided. + + As discussed in [RFC3986], URLs that specify ports other than the + default may cause problems, as may very long URLs (which are more + likely to be useful in exploiting buffer overflow bugs). + + This extension continues to use SHA-1 (as in RFC 4366) and does not + provide algorithm agility. The property required of SHA-1 in this + case is second pre-image resistance, not collision resistance. + Furthermore, even if second pre-image attacks against SHA-1 are found + in the future, an attack against client_certificate_url would require + a second pre-image that is accepted as a valid certificate by the + server and contains the same public key. + + Also note that HTTP caching proxies are common on the Internet, and + some proxies do not check for the latest version of an object + correctly. If a request using HTTP (or another caching protocol) + goes through a misconfigured or otherwise broken proxy, the proxy may + return an out-of-date response. + +11.4. Security Considerations for trusted_ca_keys + + Potentially, the CA root keys a client possesses could be regarded as + confidential information. As a result, the CA root key indication + extension should be used with care. + + The use of the SHA-1 certificate hash alternative ensures that each + certificate is specified unambiguously. This context does not + require a cryptographic hash function, so the use of SHA-1 is + considered acceptable, and no algorithm agility is provided. + +11.5. Security Considerations for truncated_hmac + + It is possible that truncated MACs are weaker than "un-truncated" + MACs. However, no significant weaknesses are currently known or + expected to exist for HMAC with MD5 or SHA-1, truncated to 80 bits. + + Note that the output length of a MAC need not be as long as the + length of a symmetric cipher key, since forging of MAC values cannot + be done off-line: in TLS, a single failed MAC guess will cause the + immediate termination of the TLS session. + + + + +Eastlake Standards Track [Page 21] + +RFC 6066 TLS Extension Definitions January 2011 + + + Since the MAC algorithm only takes effect after all handshake + messages that affect extension parameters have been authenticated by + the hashes in the Finished messages, it is not possible for an active + attacker to force negotiation of the truncated HMAC extension where + it would not otherwise be used (to the extent that the handshake + authentication is secure). Therefore, in the event that any security + problems were found with truncated HMAC in the future, if either the + client or the server for a given session were updated to take the + problem into account, it would be able to veto use of this extension. + +11.6. Security Considerations for status_request + + If a client requests an OCSP response, it must take into account that + an attacker's server using a compromised key could (and probably + would) pretend not to support the extension. In this case, a client + that requires OCSP validation of certificates SHOULD either contact + the OCSP server directly or abort the handshake. + + Use of the OCSP nonce request extension (id-pkix-ocsp-nonce) may + improve security against attacks that attempt to replay OCSP + responses; see Section 4.4.1 of [RFC2560] for further details. + +12. Normative References + + [RFC2104] Krawczyk, H., Bellare, M., and R. Canetti, "HMAC: + Keyed-Hashing for Message Authentication", RFC 2104, + February 1997. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, March 1997. + + [RFC2560] Myers, M., Ankney, R., Malpani, A., Galperin, S., and + C. Adams, "X.509 Internet Public Key Infrastructure + Online Certificate Status Protocol - OCSP", RFC 2560, + June 1999. + + [RFC2585] Housley, R. and P. Hoffman, "Internet X.509 Public Key + Infrastructure Operational Protocols: FTP and HTTP", + RFC 2585, May 1999. + + [RFC2616] Fielding, R., Gettys, J., Mogul, J., Frystyk, H., + Masinter, L., Leach, P., and T. Berners-Lee, + "Hypertext Transfer Protocol -- HTTP/1.1", RFC 2616, + June 1999. + + [RFC3986] Berners-Lee, T., Fielding, R., and L. Masinter, + "Uniform Resource Identifier (URI): Generic Syntax", + STD 66, RFC 3986, January 2005. + + + +Eastlake Standards Track [Page 22] + +RFC 6066 TLS Extension Definitions January 2011 + + + [RFC5246] Dierks, T. and E. Rescorla, "The Transport Layer + Security (TLS) Protocol Version 1.2", RFC 5246, August + 2008. + + [RFC5280] Cooper, D., Santesson, S., Farrell, S., Boeyen, S., + Housley, R., and W. Polk, "Internet X.509 Public Key + Infrastructure Certificate and Certificate Revocation + List (CRL) Profile", RFC 5280, May 2008. + + [RFC5890] Klensin, J., "Internationalized Domain Names for + Applications (IDNA): Definitions and Document + Framework", RFC 5890, August 2010. + +13. Informative References + + [RFC2712] Medvinsky, A. and M. Hur, "Addition of Kerberos Cipher + Suites to Transport Layer Security (TLS)", RFC 2712, + October 1999. + + [X509-4th] ITU-T Recommendation X.509 (2000) | ISO/IEC + 9594-8:2001, "Information Systems - Open Systems + Interconnection - The Directory: Public key and + attribute certificate frameworks". + + [X509-4th-TC1] ITU-T Recommendation X.509(2000) Corrigendum 1(2001) | + ISO/IEC 9594-8:2001/Cor.1:2002, Technical Corrigendum + 1 to ISO/IEC 9594:8:2001. + + + + + + + + + + + + + + + + + + + + + + + + +Eastlake Standards Track [Page 23] + +RFC 6066 TLS Extension Definitions January 2011 + + +Appendix A. Changes from RFC 4366 + + The significant changes between RFC 4366 and this document are + described below. + + RFC 4366 described both general extension mechanisms (for the TLS + handshake and client and server hellos) as well as specific + extensions. RFC 4366 was associated with RFC 4346, TLS 1.1. The + client and server hello extension mechanisms have been moved into RFC + 5246, TLS 1.2, so this document, which is associated with RFC 5246, + includes only the handshake extension mechanisms and the specific + extensions from RFC 4366. RFC 5246 also specifies the unknown + extension error and new extension specification considerations, so + that material has been removed from this document. + + The Server Name extension now specifies only ASCII representation, + eliminating UTF-8. It is provided that the ServerNameList can + contain more than only one name of any particular name_type. If a + server name is provided but not recognized, the server should either + continue the handshake without an error or send a fatal error. + Sending a warning-level message is not recommended because client + behavior will be unpredictable. Provision was added for the user + using the server_name extension in deciding whether or not to resume + a session. Furthermore, this extension should be the same in a + session resumption request as it was in the full handshake that + established the session. Such a resumption request must not be + accepted if the server_name extension is different, but instead a + full handshake must be done to possibly establish a new session. + + The Client Certificate URLs extension has been changed to make the + presence of a hash mandatory. + + For the case of DTLS, the requirement to report an overflow of the + negotiated maximum fragment length is made conditional on passing + authentication. + + TLS servers are now prohibited from following HTTP redirects when + retrieving certificates. + + The material was also re-organized in minor ways. For example, + information as to which errors are fatal is moved from the "Error + Alerts" section to the individual extension specifications. + + + + + + + + + +Eastlake Standards Track [Page 24] + +RFC 6066 TLS Extension Definitions January 2011 + + +Appendix B. Acknowledgements + + This document is based on material from RFC 4366 for which the + authors were S. Blake-Wilson, M. Nystrom, D. Hopwood, J. Mikkelsen, + and T. Wright. Other contributors include Joseph Salowey, Alexey + Melnikov, Peter Saint-Andre, and Adrian Farrel. + +Author's Address + + Donald Eastlake 3rd + Huawei + 155 Beaver Street + Milford, MA 01757 USA + + Phone: +1-508-333-2270 + EMail: d3e3e3@gmail.com + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Eastlake Standards Track [Page 25] + diff --git a/eval/corpora/rfc/RFC 7301 - TLS Application-Layer Protocol Negotiation Extension.txt b/eval/corpora/rfc/RFC 7301 - TLS Application-Layer Protocol Negotiation Extension.txt new file mode 100644 index 00000000..717f9f4c --- /dev/null +++ b/eval/corpora/rfc/RFC 7301 - TLS Application-Layer Protocol Negotiation Extension.txt @@ -0,0 +1,507 @@ + + + + + + +Internet Engineering Task Force (IETF) S. Friedl +Request for Comments: 7301 Cisco Systems, Inc. +Category: Standards Track A. Popov +ISSN: 2070-1721 Microsoft Corp. + A. Langley + Google Inc. + E. Stephan + Orange + July 2014 + + + Transport Layer Security (TLS) + Application-Layer Protocol Negotiation Extension + +Abstract + + This document describes a Transport Layer Security (TLS) extension + for application-layer protocol negotiation within the TLS handshake. + For instances in which multiple application protocols are supported + on the same TCP or UDP port, this extension allows the application + layer to negotiate which protocol will be used within the TLS + connection. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 5741. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + http://www.rfc-editor.org/info/rfc7301. + + + + + + + + + + + + + + + +Friedl, et al. Standards Track [Page 1] + +RFC 7301 TLS App-Layer Protocol Negotiation Ext July 2014 + + +Copyright Notice + + Copyright (c) 2014 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (http://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + +Table of Contents + + 1. Introduction . . . . . . . . . . . . . . . . . . . . . . . . 2 + 2. Requirements Language . . . . . . . . . . . . . . . . . . . . 3 + 3. Application-Layer Protocol Negotiation . . . . . . . . . . . 3 + 3.1. The Application-Layer Protocol Negotiation Extension . . 3 + 3.2. Protocol Selection . . . . . . . . . . . . . . . . . . . 5 + 4. Design Considerations . . . . . . . . . . . . . . . . . . . . 6 + 5. Security Considerations . . . . . . . . . . . . . . . . . . . 6 + 6. IANA Considerations . . . . . . . . . . . . . . . . . . . . . 7 + 7. Acknowledgements . . . . . . . . . . . . . . . . . . . . . . 8 + 8. References . . . . . . . . . . . . . . . . . . . . . . . . . 8 + 8.1. Normative References . . . . . . . . . . . . . . . . . . 8 + 8.2. Informative References . . . . . . . . . . . . . . . . . 8 + +1. Introduction + + Increasingly, application-layer protocols are encapsulated in the TLS + protocol [RFC5246]. This encapsulation enables applications to use + the existing, secure communications links already present on port 443 + across virtually the entire global IP infrastructure. + + When multiple application protocols are supported on a single server- + side port number, such as port 443, the client and the server need to + negotiate an application protocol for use with each connection. It + is desirable to accomplish this negotiation without adding network + round-trips between the client and the server, as each round-trip + will degrade an end-user's experience. Further, it would be + advantageous to allow certificate selection based on the negotiated + application protocol. + + + + + + +Friedl, et al. Standards Track [Page 2] + +RFC 7301 TLS App-Layer Protocol Negotiation Ext July 2014 + + + This document specifies a TLS extension that permits the application + layer to negotiate protocol selection within the TLS handshake. This + work was requested by the HTTPbis WG to address the negotiation of + HTTP/2 ([HTTP2]) over TLS; however, ALPN facilitates negotiation of + arbitrary application-layer protocols. + + With ALPN, the client sends the list of supported application + protocols as part of the TLS ClientHello message. The server chooses + a protocol and sends the selected protocol as part of the TLS + ServerHello message. The application protocol negotiation can thus + be accomplished within the TLS handshake, without adding network + round-trips, and allows the server to associate a different + certificate with each application protocol, if desired. + +2. Requirements Language + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "MAY", and "OPTIONAL" in this + document are to be interpreted as described in [RFC2119]. + +3. Application-Layer Protocol Negotiation + +3.1. The Application-Layer Protocol Negotiation Extension + + A new extension type ("application_layer_protocol_negotiation(16)") + is defined and MAY be included by the client in its "ClientHello" + message. + + enum { + application_layer_protocol_negotiation(16), (65535) + } ExtensionType; + + The "extension_data" field of the + ("application_layer_protocol_negotiation(16)") extension SHALL + contain a "ProtocolNameList" value. + + opaque ProtocolName<1..2^8-1>; + + struct { + ProtocolName protocol_name_list<2..2^16-1> + } ProtocolNameList; + + "ProtocolNameList" contains the list of protocols advertised by the + client, in descending order of preference. Protocols are named by + IANA-registered, opaque, non-empty byte strings, as described further + in Section 6 ("IANA Considerations") of this document. Empty strings + MUST NOT be included and byte strings MUST NOT be truncated. + + + + +Friedl, et al. Standards Track [Page 3] + +RFC 7301 TLS App-Layer Protocol Negotiation Ext July 2014 + + + Servers that receive a ClientHello containing the + "application_layer_protocol_negotiation" extension MAY return a + suitable protocol selection response to the client. The server will + ignore any protocol name that it does not recognize. A new + ServerHello extension type + ("application_layer_protocol_negotiation(16)") MAY be returned to the + client within the extended ServerHello message. The "extension_data" + field of the ("application_layer_protocol_negotiation(16)") extension + is structured the same as described above for the client + "extension_data", except that the "ProtocolNameList" MUST contain + exactly one "ProtocolName". + + Therefore, a full handshake with the + "application_layer_protocol_negotiation" extension in the ClientHello + and ServerHello messages has the following flow (contrast with + Section 7.3 of [RFC5246]): + + Client Server + + ClientHello --------> ServerHello + (ALPN extension & (ALPN extension & + list of protocols) selected protocol) + Certificate* + ServerKeyExchange* + CertificateRequest* + <-------- ServerHelloDone + Certificate* + ClientKeyExchange + CertificateVerify* + [ChangeCipherSpec] + Finished --------> + [ChangeCipherSpec] + <-------- Finished + Application Data <-------> Application Data + + Figure 1 + + * Indicates optional or situation-dependent messages that are not + always sent. + + + + + + + + + + + + +Friedl, et al. Standards Track [Page 4] + +RFC 7301 TLS App-Layer Protocol Negotiation Ext July 2014 + + + An abbreviated handshake with the + "application_layer_protocol_negotiation" extension has the following + flow: + + Client Server + + ClientHello --------> ServerHello + (ALPN extension & (ALPN extension & + list of protocols) selected protocol) + [ChangeCipherSpec] + <-------- Finished + [ChangeCipherSpec] + Finished --------> + Application Data <-------> Application Data + + Figure 2 + + Unlike many other TLS extensions, this extension does not establish + properties of the session, only of the connection. When session + resumption or session tickets [RFC5077] are used, the previous + contents of this extension are irrelevant, and only the values in the + new handshake messages are considered. + +3.2. Protocol Selection + + It is expected that a server will have a list of protocols that it + supports, in preference order, and will only select a protocol if the + client supports it. In that case, the server SHOULD select the most + highly preferred protocol that it supports and that is also + advertised by the client. In the event that the server supports no + protocols that the client advertises, then the server SHALL respond + with a fatal "no_application_protocol" alert. + + enum { + no_application_protocol(120), + (255) + } AlertDescription; + + The protocol identified in the + "application_layer_protocol_negotiation" extension type in the + ServerHello SHALL be definitive for the connection, until + renegotiated. The server SHALL NOT respond with a selected protocol + and subsequently use a different protocol for application data + exchange. + + + + + + + +Friedl, et al. Standards Track [Page 5] + +RFC 7301 TLS App-Layer Protocol Negotiation Ext July 2014 + + +4. Design Considerations + + The ALPN extension is intended to follow the typical design of TLS + protocol extensions. Specifically, the negotiation is performed + entirely within the client/server hello exchange in accordance with + the established TLS architecture. The + "application_layer_protocol_negotiation" ServerHello extension is + intended to be definitive for the connection (until the connection is + renegotiated) and is sent in plaintext to permit network elements to + provide differentiated service for the connection when the TCP or UDP + port number is not definitive for the application-layer protocol to + be used in the connection. By placing ownership of protocol + selection on the server, ALPN facilitates scenarios in which + certificate selection or connection rerouting may be based on the + negotiated protocol. + + Finally, by managing protocol selection in the clear as part of the + handshake, ALPN avoids introducing false confidence with respect to + the ability to hide the negotiated protocol in advance of + establishing the connection. If hiding the protocol is required, + then renegotiation after connection establishment, which would + provide true TLS security guarantees, would be a preferred + methodology. + +5. Security Considerations + + The ALPN extension does not impact the security of TLS session + establishment or application data exchange. ALPN serves to provide + an externally visible marker for the application-layer protocol + associated with the TLS connection. Historically, the application- + layer protocol associated with a connection could be ascertained from + the TCP or UDP port number in use. + + Implementers and document editors who intend to extend the protocol + identifier registry by adding new protocol identifiers should + consider that in TLS versions 1.2 and below the client sends these + identifiers in the clear. They should also consider that, for at + least the next decade, it is expected that browsers would normally + use these earlier versions of TLS in the initial ClientHello. + + Care must be taken when such identifiers may leak personally + identifiable information, or when such leakage may lead to profiling + or to leaking of sensitive information. If any of these apply to + this new protocol identifier, the identifier SHOULD NOT be used in + TLS configurations where it would be visible in the clear, and + documents specifying such protocol identifiers SHOULD recommend + against such unsafe use. + + + + +Friedl, et al. Standards Track [Page 6] + +RFC 7301 TLS App-Layer Protocol Negotiation Ext July 2014 + + +6. IANA Considerations + + The IANA has updated its "ExtensionType Values" registry to include + the following entry: + + 16 application_layer_protocol_negotiation + + This document establishes a registry for protocol identifiers + entitled "Application-Layer Protocol Negotiation (ALPN) Protocol IDs" + under the existing "Transport Layer Security (TLS) Extensions" + heading. + + Entries in this registry require the following fields: + + o Protocol: The name of the protocol. + o Identification Sequence: The precise set of octet values that + identifies the protocol. This could be the UTF-8 encoding + [RFC3629] of the protocol name. + o Reference: A reference to a specification that defines the + protocol. + + This registry operates under the "Expert Review" policy as defined in + [RFC5226]. The designated expert is advised to encourage the + inclusion of a reference to a permanent and readily available + specification that enables the creation of interoperable + implementations of the identified protocol. + + The initial set of registrations for this registry is as follows: + + Protocol: HTTP/1.1 + Identification Sequence: + 0x68 0x74 0x74 0x70 0x2f 0x31 0x2e 0x31 ("http/1.1") + Reference: [RFC7230] + + Protocol: SPDY/1 + Identification Sequence: + 0x73 0x70 0x64 0x79 0x2f 0x31 ("spdy/1") + Reference: + http://dev.chromium.org/spdy/spdy-protocol/spdy-protocol-draft1 + + Protocol: SPDY/2 + Identification Sequence: + 0x73 0x70 0x64 0x79 0x2f 0x32 ("spdy/2") + Reference: + http://dev.chromium.org/spdy/spdy-protocol/spdy-protocol-draft2 + + + + + + +Friedl, et al. Standards Track [Page 7] + +RFC 7301 TLS App-Layer Protocol Negotiation Ext July 2014 + + + Protocol: SPDY/3 + Identification Sequence: + 0x73 0x70 0x64 0x79 0x2f 0x33 ("spdy/3") + Reference: + http://dev.chromium.org/spdy/spdy-protocol/spdy-protocol-draft3 + +7. Acknowledgements + + This document benefitted specifically from the Next Protocol + Negotiation (NPN) extension document authored by Adam Langley and + from discussions with Tom Wesselman and Cullen Jennings, both of + Cisco. + +8. References + +8.1. Normative References + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, March 1997. + + [RFC3629] Yergeau, F., "UTF-8, a transformation format of ISO + 10646", STD 63, RFC 3629, November 2003. + + [RFC5226] Narten, T. and H. Alvestrand, "Guidelines for Writing an + IANA Considerations Section in RFCs", BCP 26, RFC 5226, + May 2008. + + [RFC5246] Dierks, T. and E. Rescorla, "The Transport Layer Security + (TLS) Protocol Version 1.2", RFC 5246, August 2008. + + [RFC7230] Fielding, R. and J. Reschke, "Hypertext Transfer Protocol + (HTTP/1.1): Message Syntax and Routing", RFC 7230, June + 2014. + +8.2. Informative References + + [HTTP2] Belshe, M., Peon, R., and M. Thomson, "Hypertext Transfer + Protocol version 2", Work in Progress, June 2014. + + [RFC5077] Salowey, J., Zhou, H., Eronen, P., and H. Tschofenig, + "Transport Layer Security (TLS) Session Resumption without + Server-Side State", RFC 5077, January 2008. + + + + + + + + + +Friedl, et al. Standards Track [Page 8] + +RFC 7301 TLS App-Layer Protocol Negotiation Ext July 2014 + + +Authors' Addresses + + Stephan Friedl + Cisco Systems, Inc. + 170 West Tasman Drive + San Jose, CA 95134 + USA + + Phone: (720)562-6785 + EMail: sfriedl@cisco.com + + + Andrei Popov + Microsoft Corp. + One Microsoft Way + Redmond, WA 98052 + USA + + EMail: andreipo@microsoft.com + + + Adam Langley + Google Inc. + USA + + EMail: agl@google.com + + + Emile Stephan + Orange + 2 avenue Pierre Marzin + Lannion F-22307 + France + + EMail: emile.stephan@orange.com + + + + + + + + + + + + + + + + +Friedl, et al. Standards Track [Page 9] + diff --git a/eval/corpora/rfc/RFC 8126 - Guidelines for Writing an IANA Considerations Section in RFCs.txt b/eval/corpora/rfc/RFC 8126 - Guidelines for Writing an IANA Considerations Section in RFCs.txt new file mode 100644 index 00000000..8db09ea1 --- /dev/null +++ b/eval/corpora/rfc/RFC 8126 - Guidelines for Writing an IANA Considerations Section in RFCs.txt @@ -0,0 +1,2635 @@ + + + + + + +Internet Engineering Task Force (IETF) M. Cotton +Request for Comments: 8126 PTI +BCP: 26 B. Leiba +Obsoletes: 5226 Huawei Technologies +Category: Best Current Practice T. Narten +ISSN: 2070-1721 IBM Corporation + June 2017 + + + Guidelines for Writing an IANA Considerations Section in RFCs + +Abstract + + Many protocols make use of points of extensibility that use constants + to identify various protocol parameters. To ensure that the values + in these fields do not have conflicting uses and to promote + interoperability, their allocations are often coordinated by a + central record keeper. For IETF protocols, that role is filled by + the Internet Assigned Numbers Authority (IANA). + + To make assignments in a given registry prudently, guidance + describing the conditions under which new values should be assigned, + as well as when and how modifications to existing values can be made, + is needed. This document defines a framework for the documentation + of these guidelines by specification authors, in order to assure that + the provided guidance for the IANA Considerations is clear and + addresses the various issues that are likely in the operation of a + registry. + + This is the third edition of this document; it obsoletes RFC 5226. + +Status of This Memo + + This memo documents an Internet Best Current Practice. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + BCPs is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + http://www.rfc-editor.org/info/rfc8126. + + + + + + + +Cotton, et al. Best Current Practice [Page 1] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + +Copyright Notice + + Copyright (c) 2017 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (http://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + +Table of Contents + + 1. Introduction . . . . . . . . . . . . . . . . . . . . . . . . 4 + 1.1. Keep IANA Considerations for IANA . . . . . . . . . . . . 4 + 1.2. For Updated Information . . . . . . . . . . . . . . . . . 5 + 1.3. A Quick Checklist Upfront . . . . . . . . . . . . . . . . 5 + 2. Creating and Revising Registries . . . . . . . . . . . . . . 7 + 2.1. Organization of Registries . . . . . . . . . . . . . . . 8 + 2.2. Documentation Requirements for Registries . . . . . . . . 8 + 2.3. Specifying Change Control for a Registry . . . . . . . . 11 + 2.4. Revising Existing Registries . . . . . . . . . . . . . . 11 + 3. Registering New Values in an Existing Registry . . . . . . . 12 + 3.1. Documentation Requirements for Registrations . . . . . . 12 + 3.2. Updating Existing Registrations . . . . . . . . . . . . . 14 + 3.3. Overriding Registration Procedures . . . . . . . . . . . 14 + 3.4. Early Allocations . . . . . . . . . . . . . . . . . . . . 15 + 4. Choosing a Registration Policy and Well-Known Policies . . . 15 + 4.1. Private Use . . . . . . . . . . . . . . . . . . . . . . . 18 + 4.2. Experimental Use . . . . . . . . . . . . . . . . . . . . 18 + 4.3. Hierarchical Allocation . . . . . . . . . . . . . . . . . 19 + 4.4. First Come First Served . . . . . . . . . . . . . . . . . 19 + 4.5. Expert Review . . . . . . . . . . . . . . . . . . . . . . 20 + 4.6. Specification Required . . . . . . . . . . . . . . . . . 21 + 4.7. RFC Required . . . . . . . . . . . . . . . . . . . . . . 22 + 4.8. IETF Review . . . . . . . . . . . . . . . . . . . . . . . 22 + 4.9. Standards Action . . . . . . . . . . . . . . . . . . . . 23 + 4.10. IESG Approval . . . . . . . . . . . . . . . . . . . . . . 23 + 4.11. Using the Well-Known Registration Policies . . . . . . . 24 + 4.12. Using Multiple Policies in Combination . . . . . . . . . 26 + 4.13. Provisional Registrations . . . . . . . . . . . . . . . . 26 + + + + + + +Cotton, et al. Best Current Practice [Page 2] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + 5. Designated Experts . . . . . . . . . . . . . . . . . . . . . 27 + 5.1. The Motivation for Designated Experts . . . . . . . . . . 27 + 5.2. The Role of the Designated Expert . . . . . . . . . . . . 27 + 5.2.1. Managing Designated Experts in the IETF . . . . . . . 29 + 5.3. Designated Expert Reviews . . . . . . . . . . . . . . . . 29 + 5.4. Expert Reviews and the Document Lifecycle . . . . . . . . 31 + 6. Well-Known Registration Status Terminology . . . . . . . . . 31 + 7. Documentation References in IANA Registries . . . . . . . . . 32 + 8. What to Do in "bis" Documents . . . . . . . . . . . . . . . . 33 + 9. Miscellaneous Issues . . . . . . . . . . . . . . . . . . . . 34 + 9.1. When There Are No IANA Actions . . . . . . . . . . . . . 34 + 9.2. Namespaces Lacking Documented Guidance . . . . . . . . . 35 + 9.3. After-the-Fact Registrations . . . . . . . . . . . . . . 35 + 9.4. Reclaiming Assigned Values . . . . . . . . . . . . . . . 35 + 9.5. Contact Person vs Assignee or Owner . . . . . . . . . . . 36 + 9.6. Closing or Obsoleting a Registry/Registrations . . . . . 37 + 10. Appeals . . . . . . . . . . . . . . . . . . . . . . . . . . . 37 + 11. Mailing Lists . . . . . . . . . . . . . . . . . . . . . . . . 37 + 12. Security Considerations . . . . . . . . . . . . . . . . . . . 37 + 13. IANA Considerations . . . . . . . . . . . . . . . . . . . . . 38 + 14. Changes Relative to Earlier Editions of BCP 26 . . . . . . . 38 + 14.1. 2016: Changes in This Document Relative to RFC 5226 . . 38 + 14.2. 2008: Changes in RFC 5226 Relative to RFC 2434 . . . . . 39 + 15. References . . . . . . . . . . . . . . . . . . . . . . . . . 40 + 15.1. Normative References . . . . . . . . . . . . . . . . . . 40 + 15.2. Informative References . . . . . . . . . . . . . . . . . 40 + Acknowledgments for This Document (2017) . . . . . . . . . . . . 46 + Acknowledgments from the Second Edition (2008) . . . . . . . . . 46 + Acknowledgments from the First Edition (1998) . . . . . . . . . . 46 + Authors' Addresses . . . . . . . . . . . . . . . . . . . . . . . 47 + + + + + + + + + + + + + + + + + + + + + +Cotton, et al. Best Current Practice [Page 3] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + +1. Introduction + + Many protocols make use of points of extensibility that use constants + to identify various protocol parameters. To ensure that the values + in these fields do not have conflicting uses and to promote + interoperability, their allocations are often coordinated by a + central record keeper. The Protocol field in the IP header [RFC791] + and MIME media types [RFC6838] are two examples of such + coordinations. + + The IETF selects an IANA Functions Operator (IFO) for protocol + parameters defined by the IETF. In the contract between the IETF and + the current IFO (ICANN), that entity is referred to as the IANA + PROTOCOL PARAMETER SERVICES Operator, or IPPSO. For consistency with + past practice, the IFO or IPPSO is referred to in this document as + "IANA" [RFC2860]. + + In this document, we call the range of possible values for such a + field a "namespace". The binding or association of a specific value + with a particular purpose within a namespace is called an assignment + (or, variously: an assigned number, assigned value, code point, + protocol constant, or protocol parameter). The act of assignment is + called a registration, and it takes place in the context of a + registry. The terms "assignment" and "registration" are used + interchangeably throughout this document. + + To make assignments in a given namespace prudently, guidance + describing the conditions under which new values should be assigned, + as well as when and how modifications to existing values can be made, + is needed. This document defines a framework for the documentation + of these guidelines by specification authors, in order to assure that + the guidance for the IANA Considerations is clear and addresses the + various issues that are likely in the operation of a registry. + + Typically, this information is recorded in a dedicated section of the + specification with the title "IANA Considerations". + +1.1. Keep IANA Considerations for IANA + + The purpose of having a dedicated IANA Considerations section is to + provide a single place to collect clear and concise information and + instructions for IANA. Technical documentation should reside in + other parts of the document; the IANA Considerations should refer to + these other sections by reference only (as needed). Using the IANA + Considerations section as primary technical documentation both hides + it from the target audience of the document and interferes with + IANA's review of the actions they need to take. + + + + +Cotton, et al. Best Current Practice [Page 4] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + An ideal IANA Considerations section clearly enumerates and specifies + each requested IANA action; includes all information IANA needs, such + as the full names of all applicable registries; and includes clear + references to elsewhere in the document for other information. + + The IANA actions are normally phrased as requests for IANA (such as, + "IANA is asked to assign the value TBD1 from the Frobozz + Registry..."); the RFC Editor will change those sentences to reflect + the actions taken ("IANA has assigned the value 83 from the Frobozz + Registry..."). + +1.2. For Updated Information + + IANA maintains a web page that includes additional clarification + information beyond what is provided here, such as minor updates and + summary guidance. Document authors should check that page. Any + significant updates to the best current practice will have to feed + into updates to BCP 26 (this document), which is definitive. + + <https://iana.org/help/protocol-registration> + +1.3. A Quick Checklist Upfront + + It's useful to be familiar with this document as a whole. But when + you return for quick reference, here are checklists for the most + common things you'll need to do and references to help with the less + common ones. + + In general... + + 1. Put all the information that IANA will need to know into the + "IANA Considerations" section of your document (see Section 1.1). + + 2. Try to keep that section only for information to IANA and to + designated expert reviewers; put significant technical + information in the appropriate technical sections of the document + (see Section 1.1). + + 3. Note that the IESG has the authority to resolve issues with IANA + registrations. If you have any questions or problems, you should + consult your document shepherd and/or working group chair, who + may ultimately involve an Area Director (see Section 3.3). + + + + + + + + + +Cotton, et al. Best Current Practice [Page 5] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + If you are creating a new registry... + + 1. Give the registry a descriptive name and provide a brief + description of its use (see Section 2.2). + + 2. Identify any registry grouping that it should be part of (see + Section 2.1). + + 3. Clearly specify what information is required in order to register + new items (see Section 2.2). Be sure to specify data types, + lengths, and valid ranges for fields. + + 4. Specify the initial set of items for the registry, if applicable + (see Section 2.2). + + 5. Make sure the change control policy for the registry is clear to + IANA, in case changes to the format or policies need to be made + later (see Sections 2.3 and 9.5). + + 6. Select a registration policy -- or a set of policies -- to use + for future registrations (see Section 4, and especially note + Sections 4.11 and 4.12). + + 7. If you're using a policy that requires a designated expert + (Expert Review or Specification Required), understand Section 5 + and provide review guidance to the designated expert (see + Section 5.3). + + 8. If any items or ranges in your registry need to be reserved for + special use or are otherwise unavailable for assignment, see + Section 6. + + If you are registering into an existing registry... + + 1. Clearly identify the registry by its exact name and optionally by + its URL (see Section 3.1). + + 2. If the registry has multiple ranges from which assignments can be + made, make it clear which range is requested (see Section 3.1). + + 3. Avoid using specific values for numeric or bit assignments, and + let IANA pick a suitable value at registration time (see + Section 3.1). This will avoid registration conflicts among + multiple documents. + + + + + + + +Cotton, et al. Best Current Practice [Page 6] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + 4. For "reference" fields, use the document that provides the best + and most current documentation for the item being registered. + Include section numbers to make it easier for readers to locate + the relevant documentation (see Sections 3.1 and 7). + + 5. Look up (in the registry's reference document) what information + is required for the registry and accurately provide all the + necessary information (see Section 3.1). + + 6. Look up (in the registry's reference document) any special rules + or processes there may be for the registry, such as posting to a + particular mailing list for comment, and be sure to follow the + process (see Section 3.1). + + 7. If the registration policy for the registry does not already + dictate the change control policy, make sure it's clear to IANA + what the change control policy is for the item, in case changes + to the registration need to be made later (see Section 9.5). + + If you're writing a "bis" document or otherwise making older + documents obsolete, see Section 8. + + If you need to make an early registration, such as for supporting + test implementations during document development, rather than waiting + for your document to be finished and approved, see [RFC7120]. + + If you need to change the format/contents or policies for an existing + registry, see Section 2.4. + + If you need to update an existing registration, see Section 3.2. + + If you need to close down a registry because it is no longer needed, + see Section 9.6. + +2. Creating and Revising Registries + + Defining a registry involves describing the namespaces to be created, + listing an initial set of assignments (if applicable), and + documenting guidelines on how future assignments are to be made. + + When defining a registry, consider structuring the namespace in such + a way that only top-level assignments need to be made with central + coordination, and those assignments can delegate lower-level + assignments so coordination for them can be distributed. This + lessens the burden on IANA for dealing with assignments, and is + particularly useful in situations where distributed coordinators have + better knowledge of their portion of the namespace and are better + suited to handling those assignments. + + + +Cotton, et al. Best Current Practice [Page 7] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + +2.1. Organization of Registries + + All registries are anchored from the IANA "Protocol Registries" page: + + <https://www.iana.org/protocols> + + That page lists registries in protocol category groups, placing + related registries together and making it easier for users of the + registries to find the necessary information. Clicking on the title + of one of the registries on the IANA Protocol Registries page will + take the reader to the details page for that registry. + + Unfortunately, we have been inconsistent in how we refer to these + entities. The group names, as they are referred to here, have been + variously called "protocol category groups", "groups", "top-level + registries", or just "registries". The registries under them have + been called "registries" or "sub-registries". + + Regardless of the terminology used, document authors should pay + attention to the registry groupings, should request that related + registries be grouped together to make related registries easier to + find, and, when creating a new registry, should check whether that + registry might best be included in an existing group. That grouping + information should be clearly communicated to IANA in the registry + creation request. + +2.2. Documentation Requirements for Registries + + Documents that create a new namespace (or modify the definition of an + existing space) and that expect IANA to play a role in maintaining + that space (serving as a repository for registered values) must + provide clear instructions on details of the namespace, either in the + IANA Considerations section or referenced from it. + + In particular, such instructions must include: + + The name of the registry + + This name will appear on the IANA web page and will be referred to + in future documents that need to allocate a value from the new + space. The full name (and abbreviation, if appropriate) should be + provided. It is highly desirable that the chosen name not be + easily confused with the name of another registry. + + When creating a registry, the group that it is a part of must be + identified using its full name, exactly as it appears in the + Protocol Registries list. + + + + +Cotton, et al. Best Current Practice [Page 8] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Providing a URL to precisely identify the registry helps IANA + understand the request. Such URLs can be removed from the RFC + prior to final publication or left in the document for reference. + If you include iana.org URLs, IANA will provide corrections, if + necessary, during their review. + + Required information for registrations + + This tells registrants what information they have to include in + their registration requests. Some registries require only the + requested value and a reference to a document where use of the + value is defined. Other registries require a more detailed + registration template that describes relevant security + considerations, internationalization considerations, and other + such information. + + Applicable registration policy + + The policy that will apply to all future requests for + registration. See Section 4. + + Size, format, and syntax of registry entries + + What fields to record in the registry, any technical requirements + on registry entries (valid ranges for integers, length limitations + on strings, and such), and the exact format in which registry + values should be displayed. For numeric assignments, one should + specify whether values are to be recorded in decimal, in + hexadecimal, or in some other format. + + Strings are expected to be ASCII, and it should be clearly + specified whether case matters, and whether, for example, strings + should be shown in the registry in uppercase or lowercase. + + Strings that represent protocol parameters will rarely, if ever, + need to contain non-ASCII characters. If non-ASCII characters are + really necessary, instructions should make it very clear that they + are allowed and that the non-ASCII characters should be + represented as Unicode characters using the "(U+XXXX)" convention. + Anyone creating such a registry should think carefully about this + and consider internationalization advice such as that in + [RFC7564], Section 10. + + + + + + + + + +Cotton, et al. Best Current Practice [Page 9] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Initial assignments and reservations + + Any initial assignments or registrations to be included. In + addition, any ranges that are to be reserved for "Private Use", + "Reserved", "Unassigned", etc. (see Section 6) should be + indicated. + + For example, a document might specify a new registry by including: + + --------------------------------------------------------------- + + X. IANA Considerations + + This document defines a new DHCP option, entitled "FooBar" (see + Section y), and assigns a value of TBD1 from the DHCP Option space + <https://www.iana.org/assignments/bootp-dhcp-parameters> + [RFC2132] [RFC2939]: + Data + Tag Name Length Meaning + ---- ---- ------ ------- + TBD1 FooBar N FooBar server + + The FooBar option also defines an 8-bit FooType field, for which + IANA is to create and maintain a new registry entitled + "FooType values" used by the FooBar option. Initial values for the + DHCP FooBar FooType registry are given below; future assignments + are to be made through Expert Review [BCP26]. Assignments consist + of a DHCP FooBar FooType name and its associated value. + + Value DHCP FooBar FooType Name Definition + ---- ------------------------ ---------- + 0 Reserved + 1 Frobnitz RFCXXXX, Section y.1 + 2 NitzFrob RFCXXXX, Section y.2 + 3-254 Unassigned + 255 Reserved + --------------------------------------------------------------- + + For examples of documents that establish registries, consult + [RFC3575], [RFC3968], and [RFC4520]. + + Any time IANA includes names and contact information in the public + registry, some individuals might prefer that their contact + information not be made public. In such cases, arrangements can be + made with IANA to keep the contact information private. + + + + + + +Cotton, et al. Best Current Practice [Page 10] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + +2.3. Specifying Change Control for a Registry + + Registry definitions and registrations within registries often need + to be changed after they are created. The process of making such + changes is complicated when it is unclear who is authorized to make + the changes. For registries created by RFCs in the IETF stream, + change control for the registry lies by default with the IETF, via + the IESG. The same is true for value registrations made in IETF- + stream RFCs. + + Because registries can be created and registrations can be made + outside the IETF stream, it can sometimes be desirable to have change + control outside the IETF and IESG, and clear specification of change + control policies is always helpful. + + It is advised, therefore, that all registries that are created + clearly specify a change control policy and a change controller. It + is also advised that registries that allow registrations from outside + the IETF stream include, for each value, the designation of a change + controller for that value. If the definition or reference for a + registered value ever needs to change, or if a registered value needs + to be deprecated, it is critical that IANA know who is authorized to + make the change. For example, the Media Types registry [RFC6838] + includes a "Change Controller" in its registration template. See + also Section 9.5. + +2.4. Revising Existing Registries + + Updating the registration process or making changes to the format of + an already existing (previously created) registry (whether created + explicitly or implicitly) follows a process similar to that used when + creating a new registry. That is, a document is produced that makes + reference to the existing namespace and then provides detailed + guidance for handling assignments in the registry or detailed + instructions about the changes required. + + If a change requires a new column in the registry, the instructions + need to be clear about how to populate that column for the existing + entries. Other changes may require similar clarity. + + Such documents are normally processed with the same document status + as the document that created the registry. Under some circumstances, + such as with a straightforward change that is clearly needed (such as + adding a "status" column), or when an earlier error needs to be + corrected, the IESG may approve an update to a registry without + requiring a new document. + + + + + +Cotton, et al. Best Current Practice [Page 11] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Example documents that updated the guidelines for assignments in + pre-existing registries include: [RFC6895], [RFC3228], and [RFC3575]. + +3. Registering New Values in an Existing Registry + +3.1. Documentation Requirements for Registrations + + Often, documents request an assignment in an existing registry (one + created by a previously published document). + + Such documents should clearly identify the registry into which each + value is to be registered. Use the exact registry name as listed on + the IANA web page, and cite the RFC where the registry is defined. + When referring to an existing registry, providing a URL to precisely + identify the registry is helpful (see Section 2.2). + + There is no need to mention what the assignment policy is when making + new assignments in existing registries, as that should be clear from + the references. However, if multiple assignment policies might + apply, as in registries with different ranges that have different + policies, it is important to make it clear which range is being + requested, so that IANA will know which policy applies and can assign + a value in the correct range. + + Be sure to provide all the information required for a registration, + and follow any special processes that are set out for the registry. + Registries sometimes require the completion of a registration + template for registration or ask registrants to post their request to + a particular mailing list for discussion prior to registration. Look + up the registry's reference document: the required information and + special processes should be documented there. + + Normally, numeric values to be used are chosen by IANA when the + document is approved; drafts should not specify final values. + Instead, placeholders such as "TBD1" and "TBD2" should be used + consistently throughout the document, giving each item to be + registered a different placeholder. The IANA Considerations should + ask the RFC Editor to replace the placeholder names with the IANA- + assigned values. When drafts need to specify numeric values for + testing or early implementations, they will either request early + allocation (see Section 3.4) or use values that have already been set + aside for testing or experimentation (if the registry in question + allows that without explicit assignment). It is important that + drafts not choose their own values, lest IANA assign one of those + values to another document in the meantime. A draft can request a + specific value in the IANA Considerations section, and IANA will + + + + + +Cotton, et al. Best Current Practice [Page 12] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + accommodate such requests when possible, but the proposed number + might have been assigned to some other use by the time the draft is + approved. + + Normally, text-string values to be used are specified in the + document, as collisions are less likely with text strings. IANA will + consult with the authors if there is, in fact, a collision, and a + different value has to be used. When drafts need to specify string + values for testing or early implementations, they sometimes use the + expected final value. But it is often useful to use a draft value + instead, possibly including the draft version number. This allows + the early implementations to be distinguished from those implementing + the final version. A document that intends to use "foobar" in the + final version might use "foobar-testing-draft-05" for the -05 version + of the draft, for example. + + For some registries, there is a long-standing policy prohibiting + assignment of names or codes on a vanity or organization-name basis. + For example, codes might always be assigned sequentially unless there + is a strong reason for making an exception. Nothing in this document + is intended to change those policies or prevent their future + application. + + As an example, the following text could be used to request assignment + of a DHCPv6 option number: + + IANA is asked to assign an option code value of TBD1 to the DNS + Recursive Name Server option and an option code value of TBD2 to + the Domain Search List option from the DHCP option code space + defined in Section 24.3 of RFC 3315. + + The IANA Considerations section should summarize all of the IANA + actions, with pointers to the relevant sections elsewhere in the + document as appropriate. Including section numbers is especially + useful when the reference document is large; the section numbers will + make it easier for those searching the reference document to find the + relevant information. + + When multiple values are requested, it is generally helpful to + include a summary table of the additions/changes. It is also helpful + for this table to be in the same format as it appears or will appear + on the IANA web site. For example: + + Value Description Reference + -------- ------------------- --------- + TBD1 Foobar this RFC, Section 3.2 + TBD2 Gumbo this RFC, Section 3.3 + TBD3 Banana this RFC, Section 3.4 + + + +Cotton, et al. Best Current Practice [Page 13] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Note: In cases where authors feel that including the full table of + changes is too verbose or repetitive, authors should still include + the table in the draft, but may include a note asking that the table + be removed prior to publication of the final RFC. + +3.2. Updating Existing Registrations + + Even after a number has been assigned, some types of registrations + contain additional information that may need to be updated over time. + + For example, MIME media types, character sets, and language tags + typically include more information than just the registered value + itself, and may need updates to items such as point-of-contact + information, security issues, pointers to updates, and literature + references. + + In such cases, the document defining the namespace must clearly state + who is responsible for maintaining and updating a registration. + Depending on the registry, it may be appropriate to specify one or + more of: + + o Letting registrants and/or nominated change controllers update + their own registrations, subject to the same constraints and + review as with new registrations. + + o Allowing attachment of comments to the registration. This can be + useful in cases where others have significant objections to a + registration, but the author does not agree to change the + registration. + + o Designating the IESG, a designated expert, or another entity as + having the right to change the registrant associated with a + registration and any requirements or conditions on doing so. This + is mainly to get around the problem when a registrant cannot be + reached in order to make necessary updates. + +3.3. Overriding Registration Procedures + + Experience has shown that the documented IANA considerations for + individual protocols do not always adequately cover the reality of + registry operation or are not sufficiently clear. In addition, + documented IANA considerations are sometimes found to be too + stringent to allow even working group documents (for which there is + strong consensus) to perform a registration in advance of actual RFC + publication. + + + + + + +Cotton, et al. Best Current Practice [Page 14] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + In order to allow assignments in such cases, the IESG is granted + authority to override registration procedures and approve assignments + on a case-by-case basis. + + The intention here is not to overrule properly documented procedures + or to obviate the need for protocols to properly document their IANA + considerations. Rather, it is to permit assignments in specific + cases where it is obvious that the assignment should just be made, + but updating the IANA process beforehand is too onerous. + + When the IESG is required to take action as described above, it is a + strong indicator that the applicable registration procedures should + be updated, possibly in parallel with the work that instigated it. + + IANA always has the discretion to ask the IESG for advice or + intervention when they feel it is needed, such as in cases where + policies or procedures are unclear to them, where they encounter + issues or questions they are unable to resolve, or where registration + requests or patterns of requests appear to be unusual or abusive. + +3.4. Early Allocations + + IANA normally takes its actions when a document is approved for + publication. There are times, though, when early allocation of a + value is important for the development of a technology, for example, + when early implementations are created while the document is still + under development. + + IANA has a mechanism for handling such early allocations in some + cases. See [RFC7120] for details. It is usually not necessary to + explicitly mark a registry as allowing early allocation, because the + general rules will apply. + +4. Choosing a Registration Policy and Well-Known Policies + + A registration policy is the policy that controls how new assignments + in a registry are accepted. There are several issues to consider + when defining the registration policy. + + If the registry's namespace is limited, assignments will need to be + made carefully to prevent exhaustion. + + + + + + + + + + +Cotton, et al. Best Current Practice [Page 15] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Even when the space is essentially unlimited, it is still often + desirable to have at least a minimal review prior to assignment in + order to: + + o prevent the hoarding of or unnecessary wasting of values. For + example, if the space consists of text strings, it may be + desirable to prevent entities from obtaining large sets of strings + that correspond to desirable names (existing company names, for + example). + + o provide a sanity check that the request actually makes sense and + is necessary. Experience has shown that some level of minimal + review from a subject matter expert is useful to prevent + assignments in cases where the request is malformed or not + actually needed (for example, an existing assignment for an + essentially equivalent service already exists). + + Perhaps most importantly, unreviewed extensions can impact + interoperability and security. See [RFC6709]. + + When the namespace is essentially unlimited and there are no + potential interoperability or security issues, assigned numbers can + usually be given out to anyone without any subjective review. In + such cases, IANA can make assignments directly, provided that IANA is + given detailed instructions on what types of requests it should + grant, and it is able to do so without exercising subjective + judgment. + + When this is not the case, some level of review is required. + However, it's important to balance adequate review and ease of + registration. In many cases, those making registrations will not be + IETF participants; requests often come from other standards + organizations, from organizations not directly involved in standards, + from ad-hoc community work (from an open-source project, for + example), and so on. Registration must not be unnecessarily + difficult, unnecessarily costly (in terms of time and other + resources), nor unnecessarily subject to denial. + + While it is sometimes necessary to restrict what gets registered + (e.g., for limited resources such as bits in a byte, or for items for + which unsupported values can be damaging to protocol operation), in + many cases having what's in use represented in the registry is more + important. Overly strict review criteria and excessive cost (in time + and effort) discourage people from even attempting to make a + registration. If a registry fails to reflect the protocol elements + actually in use, it can adversely affect deployment of protocols on + the Internet, and the registry itself is devalued. + + + + +Cotton, et al. Best Current Practice [Page 16] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Therefore, it is important to think specifically about the + registration policy, and not just pick one arbitrarily nor copy text + from another document. Working groups and other document developers + should use care in selecting appropriate registration policies when + their documents create registries. They should select the least + strict policy that suits a registry's needs, and look for specific + justification for policies that require significant community + involvement (those stricter than Expert Review or Specification + Required, in terms of the well-known policies). The needs here will + vary from registry to registry, and, indeed, over time, and this BCP + will not be the last word on the subject. + + The following policies are defined for common usage. These cover a + range of typical policies that have been used to describe the + procedures for assigning new values in a namespace. It is not + strictly required that documents use these terms; the actual + requirement is that the instructions to IANA be clear and + unambiguous. However, use of these terms is strongly recommended + because their meanings are widely understood. Newly minted policies, + including ones that combine the elements of procedures associated + with these terms in novel ways, may be used if none of these policies + are suitable; it will help the review process if an explanation is + included as to why that is the case. The terms are fully explained + in the following subsections. + + 1. Private Use + 2. Experimental Use + 3. Hierarchical Allocation + 4. First Come First Served + 5. Expert Review + 6. Specification Required + 7. RFC Required + 8. IETF Review + 9. Standards Action + 10. IESG Approval + + It should be noted that it often makes sense to partition a namespace + into multiple categories, with assignments within each category + handled differently. Many protocols now partition namespaces into + two or more parts, with one range reserved for Private or + Experimental Use while other ranges are reserved for globally unique + assignments assigned following some review process. Dividing a + namespace into ranges makes it possible to have different policies in + place for different ranges and different use cases. + + Similarly, it will often be useful to specify multiple policies in + parallel, with each policy being used under different circumstances. + For more discussion of that topic, see Section 4.12. + + + +Cotton, et al. Best Current Practice [Page 17] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Examples of RFCs that specify multiple policies in parallel: + + LDAP [RFC4520] + TLS ClientCertificateType Identifiers [RFC5246] (as detailed in + the subsections below) + MPLS Pseudowire Types Registry [RFC4446] + +4.1. Private Use + + Private Use is for private or local use only, with the type and + purpose defined by the local site. No attempt is made to prevent + multiple sites from using the same value in different (and + incompatible) ways. IANA does not record assignments from registries + or ranges with this policy (and therefore there is no need for IANA + to review them) and assignments are not generally useful for broad + interoperability. It is the responsibility of the sites making use + of the Private Use range to ensure that no conflicts occur (within + the intended scope of use). + + Examples: + + Site-specific options in DHCP [RFC2939] + Fibre Channel Port Type Registry [RFC4044] + TLS ClientCertificateType Identifiers 224-255 [RFC5246] + +4.2. Experimental Use + + Experimental Use is similar to Private Use, but with the purpose + being to facilitate experimentation. See [RFC3692] for details. + IANA does not record assignments from registries or ranges with this + policy (and therefore there is no need for IANA to review them) and + assignments are not generally useful for broad interoperability. + Unless the registry explicitly allows it, it is not appropriate for + documents to select explicit values from registries or ranges with + this policy. Specific experiments will select a value to use during + the experiment. + + When code points are set aside for Experimental Use, it's important + to make clear any expected restrictions on experimental scope. For + example, say whether it's acceptable to run experiments using those + code points over the open Internet or whether such experiments should + be confined to more closed environments. See [RFC6994] for an + example of such considerations. + + Example: + + Experimental Values in IPv4, IPv6, ICMPv4, ICMPv6, UDP, and TCP + Headers [RFC4727] + + + +Cotton, et al. Best Current Practice [Page 18] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + +4.3. Hierarchical Allocation + + With Hierarchical Allocation, delegated administrators are given + control over part of the namespace and can assign values in that part + of the namespace. IANA makes allocations in the higher levels of the + namespace according to one of the other policies. + + Examples: + + o DNS names - IANA manages the top-level domains (TLDs), and, as + [RFC1591] says: + + Under each TLD may be created a hierarchy of names. Generally, + under the generic TLDs the structure is very flat. That is, + many organizations are registered directly under the TLD, and + any further structure is up to the individual organizations. + + o Object Identifiers - defined by ITU-T recommendation X.208. + According to <http://www.alvestrand.no/objectid/>, some registries + include + + * IANA, which hands out OIDs under the "Private Enterprises" + branch, + * ANSI, which hands out OIDs under the "US Organizations" branch, + and + * BSI, which hands out OIDs under the "UK Organizations" branch. + + o URN namespaces - IANA registers URN Namespace IDs (NIDs + [RFC8141]), and the organization registering an NID is responsible + for allocations of URNs within that namespace. + +4.4. First Come First Served + + For the First Come First Served policy, assignments are made to + anyone on a first come, first served basis. There is no substantive + review of the request, other than to ensure that it is well-formed + and doesn't duplicate an existing assignment. However, requests must + include a minimal amount of clerical information, such as a point of + contact (including an email address, and sometimes a postal address) + and a brief description of how the value will be used. Additional + information specific to the type of value requested may also need to + be provided, as defined by the namespace. For numbers, IANA + generally assigns the next in-sequence unallocated value, but other + values may be requested and assigned if an extenuating circumstance + exists. With names, specific text strings can usually be requested. + + + + + + +Cotton, et al. Best Current Practice [Page 19] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + When creating a new registry with First Come First Served as the + registration policy, in addition to the contact person field or + reference, the registry should contain a field for change controller. + Having a change controller for each entry for these types of + registrations makes authorization of future modifications more clear. + See Section 2.3. + + It is important that changes to the registration of a First Come + First Served code point retain compatibility with the current usage + of that code point, so changes need to be made with care. The change + controller should not, in most cases, be requesting incompatible + changes nor repurposing a registered code point. See also Sections + 9.4 and 9.5. + + A working group or any other entity that is developing a protocol + based on a First Come First Served code point has to be extremely + careful that the protocol retains wire compatibility with current use + of the code point. Once that is no longer true, the new work needs + to change to a different code point (and register that use at the + appropriate time). + + It is also important to understand that First Come First Served + really has no filtering. Essentially, any well-formed request is + accepted. + + Examples: + + SASL mechanism names [RFC4422] + LDAP Protocol Mechanisms and LDAP Syntax [RFC4520] + +4.5. Expert Review + + For the Expert Review policy, review and approval by a designated + expert (see Section 5) is required. While this does not necessarily + require formal documentation, information needs to be provided with + the request for the designated expert to evaluate. The registry's + definition needs to make clear to registrants what information is + necessary. The actual process for requesting registrations is + administered by IANA (see Section 1.2 for details). + + (This policy was also called "Designated Expert" in earlier editions + of this document. The current term is "Expert Review".) + + The required documentation and review criteria, giving clear guidance + to the designated expert, should be provided when defining the + registry. It is particularly important to lay out what should be + considered when performing an evaluation and reasons for rejecting a + request. It is also a good idea to include, when possible, a sense + + + +Cotton, et al. Best Current Practice [Page 20] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + of whether many registrations are expected over time, or if the + registry is expected to be updated infrequently or in exceptional + circumstances only. + + Thorough understanding of Section 5 is important when deciding on an + Expert Review policy and designing the guidance to the designated + expert. + + Good examples of guidance to designated experts: + + Extensible Authentication Protocol (EAP) [RFC3748], Sections 6 and + 7.2 + North-Bound Distribution of Link-State and TE Information Using + BGP [RFC7752], Section 5.1 + + When creating a new registry with Expert Review as the registration + policy, in addition to the contact person field or reference, the + registry should contain a field for change controller. Having a + change controller for each entry for these types of registrations + makes authorization of future modifications more clear. See + Section 2.3. + + Examples: + + EAP Method Types [RFC3748] + HTTP Digest AKA algorithm versions [RFC4169] + URI schemes [RFC7595] + GEOPRIV Location Types [RFC4589] + +4.6. Specification Required + + For the Specification Required policy, review and approval by a + designated expert (see Section 5) is required, and the values and + their meanings must be documented in a permanent and readily + available public specification, in sufficient detail so that + interoperability between independent implementations is possible. + This policy is the same as Expert Review, with the additional + requirement of a formal public specification. In addition to the + normal review of such a request, the designated expert will review + the public specification and evaluate whether it is sufficiently + stable and permanent, and sufficiently clear and technically sound to + allow interoperable implementations. + + The intention behind "permanent and readily available" is that a + document can reasonably be expected to be findable and retrievable + long after IANA assignment of the requested value. Publication of an + RFC is an ideal means of achieving this requirement, but + Specification Required is intended to also cover the case of a + + + +Cotton, et al. Best Current Practice [Page 21] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + document published outside of the RFC path, including informal + documentation. + + For RFC publication, formal review by the designated expert is still + requested, but the normal RFC review process is expected to provide + the necessary review for interoperability. The designated expert's + review is still important, but it's equally important to note that + when there is IETF consensus, the expert can sometimes be "in the + rough" (see also the last paragraph of Section 5.4). + + As with Expert Review (Section 4.5), clear guidance to the designated + expert should be provided when defining the registry, and thorough + understanding of Section 5 is important. + + When specifying this policy, just use the term "Specification + Required". Some specifications have chosen to refer to it as "Expert + Review with Specification Required", and that only causes confusion. + + Examples: + + Diffserv-aware TE Bandwidth Constraints Model Identifiers + [RFC4124] + TLS ClientCertificateType Identifiers 64-223 [RFC5246] + ROHC Profile Identifiers [RFC5795] + +4.7. RFC Required + + With the RFC Required policy, the registration request, along with + associated documentation, must be published in an RFC. The RFC need + not be in the IETF stream, but may be in any RFC stream (currently an + RFC may be in the IETF, IRTF, IAB, or Independent Submission streams + [RFC5742]). + + Unless otherwise specified, any type of RFC is sufficient (currently + Standards Track, BCP, Informational, Experimental, or Historic). + + Examples: + + DNSSEC DNS Security Algorithm Numbers [RFC6014] + Media Control Channel Framework registries [RFC6230] + DANE TLSA Certificate Usages [RFC6698] + +4.8. IETF Review + + (Formerly called "IETF Consensus" in the first edition of this + document.) With the IETF Review policy, new values are assigned only + through RFCs in the IETF Stream -- those that have been shepherded + through the IESG as AD-Sponsored or IETF working group documents + + + +Cotton, et al. Best Current Practice [Page 22] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + [RFC2026] [RFC5378], have gone through IETF Last Call, and have been + approved by the IESG as having IETF consensus. + + The intent is that the document and proposed assignment will be + reviewed by the IETF community (including appropriate IETF working + groups, directorates, and other experts) and by the IESG, to ensure + that the proposed assignment will not negatively affect + interoperability or otherwise extend IETF protocols in an + inappropriate or damaging manner. + + Unless otherwise specified, any type of RFC is sufficient (currently + Standards Track, BCP, Informational, Experimental, or Historic). + + Examples: + + IPSECKEY Algorithm Types [RFC4025] + TLS Extension Types [RFC5246] + +4.9. Standards Action + + For the Standards Action policy, values are assigned only through + Standards Track or Best Current Practice RFCs in the IETF Stream. + + Examples: + + BGP message types [RFC4271] + Mobile Node Identifier option types [RFC4283] + TLS ClientCertificateType Identifiers 0-63 [RFC5246] + DCCP Packet Types [RFC4340] + +4.10. IESG Approval + + New assignments may be approved by the IESG. Although there is no + requirement that the request be documented in an RFC, the IESG has + the discretion to request documents or other supporting materials on + a case-by-case basis. + + IESG Approval is not intended to be used often or as a "common case"; + indeed, it has seldom been used in practice. Rather, it is intended + to be available in conjunction with other policies as a fall-back + mechanism in the case where one of the other allowable approval + mechanisms cannot be employed in a timely fashion or for some other + compelling reason. IESG Approval is not intended to circumvent the + public review processes implied by other policies that could have + been employed for a particular assignment. IESG Approval would be + appropriate, however, in cases where expediency is desired and there + is strong consensus (such as from a working group) for making the + assignment. + + + +Cotton, et al. Best Current Practice [Page 23] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Before approving a request, the IESG might consider consulting the + community, via a "call for comments" that provides as much + information as is reasonably possible about the request. + + Examples: + + IPv4 Multicast address assignments [RFC5771] + IPv4 IGMP Type and Code values [RFC3228] + Mobile IPv6 Mobility Header Type and Option values [RFC6275] + +4.11. Using the Well-Known Registration Policies + + Because the well-known policies benefit from both community + experience and wide understanding, their use is encouraged, and the + creation of new policies needs to be accompanied by reasonable + justification. + + It is also acceptable to cite one or more well-known policies and + include additional guidelines for what kind of considerations should + be taken into account by the review process. + + For example, for media-type registrations [RFC6838], a number of + different situations are covered that involve the use of IETF Review + and Specification Required, while also including specific additional + criteria the designated expert should follow. This is not meant to + represent a registration procedure, but to show an example of what + can be done when special circumstances need to be covered. + + The well-known policies from "First Come First Served" to "Standards + Action" specify a range of policies in increasing order of strictness + (using the numbering from the full list in Section 4): + + 4. First Come First Served + No review, minimal documentation. + + 5 and 6 (of equal strictness). + + 5. Expert Review + Expert review with sufficient documentation for review. + + 6. Specification Required + Significant stable public documentation sufficient for + interoperability. + + 7. RFC Required + Any RFC publication, IETF or a non-IETF Stream. + + 8. IETF Review + + + +Cotton, et al. Best Current Practice [Page 24] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + RFC publication, IETF Stream only, but need not be Standards + Track. + + 9. Standards Action + RFC publication, IETF Stream, Standards Track or BCP only. + + Examples of situations that might merit IETF Review or Standards + Action include the following: + + o When a resource is limited, such as bits in a byte (or in two + bytes, or four), or numbers in a limited range. In these cases, + allowing registrations that haven't been carefully reviewed and + agreed to by community consensus could too quickly deplete the + allowable values. + + o When thorough community review is necessary to avoid extending or + modifying the protocol in ways that could be damaging. One + example is in defining new command codes, as opposed to options + that use existing command codes: the former might require a strict + policy, where a more relaxed policy could be adequate for the + latter. Another example is in defining protocol elements that + change the semantics of existing operations. + + o When there are security implications with respect to the resource, + and thorough review is needed to ensure that the new usage is + sound. Examples of this include lists of acceptable hashing and + cryptographic algorithms, and assignment of transport ports in the + system range. + + When reviewing a document that asks IANA to create a new registry or + change a registration policy to any policy more stringent than Expert + Review or Specification Required, the IESG should ask for + justification to ensure that more relaxed policies have been + considered and that the more strict policy is the right one. + + Accordingly, document developers need to anticipate this and document + their considerations for selecting the specified policy (ideally, in + the document itself; failing that, in the shepherd writeup). + Likewise, the document shepherd should ensure that the selected + policies have been justified before sending the document to the IESG. + + When specifications are revised, registration policies should be + reviewed in light of experience since the policies were set. + +4.12. Using Multiple Policies in Combination + + In some situations, it is necessary to define multiple registration + policies. For example, registrations through the normal IETF process + + + +Cotton, et al. Best Current Practice [Page 25] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + might use one policy, while registrations from outside the process + would have a different policy applied. + + Thus, a particular registry might want to use a policy such as "RFC + Required" or "IETF Review" sometimes, with a designated expert + checking a "Specification Required" policy at other times. + + The alternative to using a combination requires either that all + requests come through RFCs or that requests in RFCs go through review + by the designated expert, even though they already have IETF review + and consensus. + + This can be documented in the IANA Considerations section when the + registry is created, for example: + + IANA is asked to create the registry "Fruit Access Flags" under + the "Fruit Parameters" group. New registrations will be permitted + through either the IETF Review policy or the Specification + Required policy [BCP26]. The latter should be used only for + registrations requested by SDOs outside the IETF. Registrations + requested in IETF documents will be subject to IETF review. + + Such combinations will commonly use one of {Standards Action, IETF + Review, RFC Required} in combination with one of {Specification + Required, Expert Review}. Guidance should be provided about when + each policy is appropriate, as in the example above. + +4.13. Provisional Registrations + + Some existing registries have policies that allow provisional + registration: see URI Schemes [RFC7595] and Email Header Fields + [RFC3864]. Registrations that are designated as provisional are + usually defined as being more readily created, changed, reassigned, + moved to another status, or removed entirely. URI Schemes, for + example, allow provisional registrations to be made with incomplete + information. + + Allowing provisional registration ensures that the primary goal of + maintaining a registry -- avoiding collisions between incompatible + semantics -- is achieved without the side effect of "endorsing" the + protocol mechanism the provisional value is used for. Provisional + registrations for codepoints that are ultimately standardized can be + promoted to permanent status. The criteria that are defined for + converting a provisional registration to permanent will likely be + more strict than those that allowed the provisional registration. + + If your registry does not have a practical limit on codepoints, + perhaps adding the option for provisional registrations might be + + + +Cotton, et al. Best Current Practice [Page 26] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + right for that registry as well. + +5. Designated Experts + +5.1. The Motivation for Designated Experts + + Discussion on a mailing list can provide valuable technical feedback, + but opinions often vary and discussions may continue for some time + without clear resolution. In addition, IANA cannot participate in + all of these mailing lists and cannot determine if or when such + discussions reach consensus. Therefore, IANA relies on a "designated + expert" for advice regarding the specific question of whether an + assignment should be made. The designated expert is an individual + who is responsible for carrying out an appropriate evaluation and + returning a recommendation to IANA. + + It should be noted that a key motivation for having designated + experts is for the IETF to provide IANA with a subject matter expert + to whom the evaluation process can be delegated. IANA forwards + requests for an assignment to the expert for evaluation, and the + expert (after performing the evaluation) informs IANA as to whether + or not to make the assignment or registration. In most cases, the + registrants do not work directly with the designated experts. The + list of designated experts for a registry is listed in the registry. + + It will often be useful to use a designated expert only some of the + time, as a supplement to other processes. For more discussion of + that topic, see Section 4.12. + +5.2. The Role of the Designated Expert + + The designated expert is responsible for coordinating the appropriate + review of an assignment request. The review may be wide or narrow, + depending on the situation and the judgment of the designated expert. + This may involve consultation with a set of technology experts, + discussion on a public mailing list, consultation with a working + group (or its mailing list if the working group has disbanded), etc. + Ideally, the designated expert follows specific review criteria as + documented with the protocol that creates or uses the namespace. See + the IANA Considerations sections of [RFC3748] and [RFC3575] for + specific examples. + + Designated experts are expected to be able to defend their decisions + to the IETF community, and the evaluation process is not intended to + be secretive or bestow unquestioned power on the expert. Experts are + expected to apply applicable documented review or vetting procedures, + or in the absence of documented criteria, follow generally accepted + norms such as those in Section 5.3. Designated experts are generally + + + +Cotton, et al. Best Current Practice [Page 27] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + not expected to be "gatekeepers", setting out to make registrations + difficult to obtain, unless the guidance in the defining document + specifies that they should act as such. Absent stronger guidance, + the experts should be evaluating registration requests for + completeness, interoperability, and conflicts with existing protocols + and options. + + It has proven useful to have multiple designated experts for some + registries. Sometimes those experts work together in evaluating a + request, while in other cases additional experts serve as backups, + acting only when the primary expert is unavailable. In registries + with a pool of experts, the pool often has a single chair responsible + for defining how requests are to be assigned to and reviewed by + experts. In other cases, IANA might assign requests to individual + members in sequential or approximate random order. The document + defining the registry can, if it's appropriate for the situation, + specify how the group should work -- for example, it might be + appropriate to specify rough consensus on a mailing list, within a + related working group or among a pool of designated experts. + + In cases of disagreement among multiple experts, it is the + responsibility of those experts to make a single clear recommendation + to IANA. It is not appropriate for IANA to resolve disputes among + experts. In extreme situations, such as deadlock, the designating + body may need to step in to resolve the problem. + + If a designated expert has a conflict of interest for a particular + review (is, for example, an author or significant proponent of a + specification related to the registration under review), that expert + should recuse himself. In the event that all the designated experts + are conflicted, they should ask that a temporary expert be designated + for the conflicted review. The responsible AD may then appoint + someone or the AD may handle the review. + + This document defines the designated expert mechanism with respect to + documents in the IETF stream only. If other streams want to use + registration policies that require designated experts, it is up to + those streams (or those documents) to specify how those designated + experts are appointed and managed. What is described below, with + management by the IESG, is only appropriate for the IETF stream. + +5.2.1. Managing Designated Experts in the IETF + + Designated experts for registries created by the IETF are appointed + by the IESG, normally upon recommendation by the relevant Area + Director. They may be appointed at the time a document creating or + updating a namespace is approved by the IESG, or subsequently, when + the first registration request is received. Because experts + + + +Cotton, et al. Best Current Practice [Page 28] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + originally appointed may later become unavailable, the IESG will + appoint replacements as necessary. The IESG may remove any + designated expert that it appointed, at its discretion. + + The normal appeals process, as described in [RFC2026], Section 6.5.1, + applies to issues that arise with the designated expert team. For + this purpose, the designated expert team takes the place of the + working group in that description. + +5.3. Designated Expert Reviews + + In the years since [RFC2434] was published and put to use, experience + has led to the following observations: + + o A designated expert must respond in a timely fashion, normally + within a week for simple requests to a few weeks for more complex + ones. Unreasonable delays can cause significant problems for + those needing assignments, such as when products need code points + to ship. This is not to say that all reviews can be completed + under a firm deadline, but they must be started, and the requester + and IANA should have some transparency into the process if an + answer cannot be given quickly. + + o If a designated expert does not respond to IANA's requests within + a reasonable period of time, either with a response or with a + reasonable explanation for the delay (some requests may be + particularly complex), and if this is a recurring event, IANA must + raise the issue with the IESG. Because of the problems caused by + delayed evaluations and assignments, the IESG should take + appropriate actions to ensure that the expert understands and + accepts his or her responsibilities, or appoint a new expert. + + o The designated expert is not required to personally bear the + burden of evaluating and deciding all requests, but acts as a + shepherd for the request, enlisting the help of others as + appropriate. In the case that a request is denied, and rejecting + the request is likely to be controversial, the expert should have + the support of other subject matter experts. That is, the expert + must be able to defend a decision to the community as a whole. + + When a designated expert is used, the documentation should give clear + guidance to the designated expert, laying out criteria for performing + an evaluation and reasons for rejecting a request. In the case where + there are no specific documented criteria, the presumption should be + that a code point should be granted unless there is a compelling + reason to the contrary (and see also Section 5.4). Reasons that have + been used to deny requests have included these: + + + + +Cotton, et al. Best Current Practice [Page 29] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + o Scarcity of code points, where the finite remaining code points + should be prudently managed, or where a request for a large number + of code points is made and a single code point is the norm. + + o Documentation is not of sufficient clarity to evaluate or ensure + interoperability. + + o The code point is needed for a protocol extension, but the + extension is not consistent with the documented (or generally + understood) architecture of the base protocol being extended and + would be harmful to the protocol if widely deployed. It is not + the intent that "inconsistencies" refer to minor differences "of a + personal preference nature". Instead, they refer to significant + differences such as inconsistencies with the underlying security + model, implying a change to the semantics of an existing message + type or operation, requiring unwarranted changes in deployed + systems (compared with alternate ways of achieving a similar + result), etc. + + o The extension would cause problems with existing deployed systems. + + o The extension would conflict with one under active development by + the IETF, and having both would harm rather than foster + interoperability. + + Documents must not name the designated expert(s) in the document + itself; instead, any suggested names should be relayed to the + appropriate Area Director at the time the document is sent to the + IESG for approval. This is usually done in the document shepherd + writeup. + + If the request should also be reviewed on a specific public mailing + list, its address should be specified. + + + + + + + + + + + + + + + + + + +Cotton, et al. Best Current Practice [Page 30] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + +5.4. Expert Reviews and the Document Lifecycle + + Review by the designated expert is necessarily done at a particular + point in time and represents review of a particular version of the + document. While reviews are generally done around the time of IETF + Last Call, deciding when the review should take place is a question + of good judgment. And while rereviews might be done when it's + acknowledged that the documentation of the registered item has + changed substantially, making sure that rereview happens requires + attention and care. + + It is possible, through carelessness, accident, inattentiveness, or + even willful disregard, that changes might be made after the + designated expert's review and approval that would, if the document + were rereviewed, cause the expert not to approve the registration. + It is up to the IESG, with the token held by the responsible Area + Director, to be alert to such situations and to recognize that such + changes need to be checked. + + For registrations made from documents on the Standards Track, there + is often expert review required (by the registration policy) in + addition to IETF consensus (for approval as a Standards Track RFC). + In such cases, the review by the designated expert needs to be + timely, submitted before the IESG evaluates the document. The IESG + should generally not hold the document up waiting for a late review. + It is also not intended for the expert review to override IETF + consensus: the IESG should consider the review in its own evaluation, + as it would do for other Last Call reviews. + +6. Well-Known Registration Status Terminology + + The following labels describe the status of an assignment or range of + assignments: + + Private Use: Private use only (not assigned), as described in + Section 4.1. + + Experimental: Available for general experimental use as described + in [RFC3692]. IANA does not record specific assignments for + any particular use. + + Unassigned: Not currently assigned, and available for assignment + via documented procedures. While it's generally clear that + any values that are not registered are unassigned and + available for assignment, it is sometimes useful to + explicitly specify that situation. Note that this is + distinctly different from "Reserved". + + + + +Cotton, et al. Best Current Practice [Page 31] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Reserved: Not assigned and not available for assignment. + Reserved values are held for special uses, such as to extend + the namespace when it becomes exhausted. "Reserved" is also + sometimes used to designate values that had been assigned + but are no longer in use, keeping them set aside as long as + other unassigned values are available. Note that this is + distinctly different from "Unassigned". + + Reserved values can be released for assignment by the change + controller for the registry (this is often the IESG, for + registries created by RFCs in the IETF stream). + + Known Unregistered Use: It's known that the assignment or range + is in use without having been defined in accordance with + reasonable practice. Documentation for use of the + assignment or range may be unavailable, inadequate, or + conflicting. This is a warning against use, as well as an + alert to network operators who might see these values in use + on their networks. + +7. Documentation References in IANA Registries + + Usually, registries and registry entries include references to + documentation (RFCs or other documents). The purpose of these + references is to provide pointers for implementors to find details + necessary for implementation, NOT to simply note what document + created the registry or entry. Therefore: + + o If a document registers an item that is defined and explained + elsewhere, the registered reference should be to the document + containing the definition, not to the document that is merely + performing the registration. + + o If the registered item is defined and explained in the current + document, it is important to include sufficient information to + enable implementors to understand the item and to create a proper + implementation. + + o If the registered item is explained primarily in a specific + section of the reference document, it is useful to include a + section reference. For example, "[RFC4637], Section 3.2", rather + than just "[RFC4637]". + + o For documentation of a new registry, the reference should provide + information about the registry itself, not just a pointer to the + creation of it. Useful information includes the purpose of the + registry, a rationale for its creation, documentation of the + process and policy for new registrations, guidelines for new + + + +Cotton, et al. Best Current Practice [Page 32] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + registrants or designated experts, and other such related + information. But note that, while it's important to include this + information in the document, it needn't all be in the IANA + Considerations section. See Section 1.1. + +8. What to Do in "bis" Documents + + On occasion, an RFC is issued that obsoletes a previous edition of + the same document. We sometimes call these "bis" documents, such as + when RFC 4637 is to be obsoleted by draft-ietf-foo-rfc4637bis. When + the original document created registries and/or registered entries, + there is a question of how to handle the IANA Considerations section + in the "bis" document. + + If the registrations specify the original document as a reference, + those registrations should be updated to point to the current (not + obsolete) documentation for those items. Usually, that will mean + changing the reference to be the "bis" document. + + There will, though, be times when a document updates another, but + does not make it obsolete, and the definitive reference is changed + for some items but not for others. Be sure that the references + always point to the correct, current documentation for each item. + + For example, suppose RFC 4637 registered the "BANANA" flag in the + "Fruit Access Flags" registry, and the documentation for that flag is + in Section 3.2. + + The current registry might look, in part, like this: + + Name Description Reference + -------- ------------------- --------- + BANANA Flag for bananas [RFC4637], Section 3.2 + + If draft-ietf-foo-rfc4637bis obsoletes RFC 4637 and, because of some + rearrangement, now documents the flag in Section 4.1.2, the IANA + Considerations of the bis document might contain text such as this: + + IANA is asked to change the registration information for the + BANANA flag in the "Fruit Access Flags" registry to the + following: + + Name Description Reference + -------- ------------------- --------- + BANANA Flag for bananas [[this RFC]], Section 4.2.1 + + + + + + +Cotton, et al. Best Current Practice [Page 33] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + In many cases, if there are a number of registered references to the + original RFC and the document organization has not changed the + registered section numbering much, it may simply be reasonable to do + this: + + Because this document obsoletes RFC 4637, IANA is asked to change + all registration information that references [RFC4637] to instead + reference [[this RFC]]. + + If information for registered items has been or is being moved to + other documents, then the registration information should be changed + to point to those other documents. In most cases, documentation + references should not be left pointing to the obsoleted document for + registries or registered items that are still in current use. For + registries or registered items that are no longer in current use, it + will usually make sense to leave the references pointing to the old + document -- the last current reference for the obsolete items. The + main point is to make sure that the reference pointers are as useful + and current as is reasonable, and authors should consider that as + they write the IANA Considerations for the new document. As always: + do the right thing, and there is flexibility to allow for that. + + It is extremely important to be clear in your instructions regarding + updating references, especially in cases where some references need + to be updated and others do not. + +9. Miscellaneous Issues + +9.1. When There Are No IANA Actions + + Before an Internet-Draft can be published as an RFC, IANA needs to + know what actions (if any) it needs to perform. Experience has shown + that it is not always immediately obvious whether a document has no + IANA actions, without reviewing the document in some detail. In + order to make it clear to IANA that it has no actions to perform (and + that the author has consciously made such a determination), such + documents should, after the authors confirm that this is the case, + include an IANA Considerations section that states: + + This document has no IANA actions. + + IANA prefers that these "empty" IANA Considerations sections be left + in the document for the record: it makes it clear later on that the + document explicitly said that no IANA actions were needed (and that + it wasn't just omitted). This is a change from the prior practice of + requesting that such sections be removed by the RFC Editor, and + authors are asked to accommodate this change. + + + + +Cotton, et al. Best Current Practice [Page 34] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + +9.2. Namespaces Lacking Documented Guidance + + For all existing RFCs that either explicitly or implicitly rely on + IANA to make assignments without specifying a precise assignment + policy, IANA will work with the IESG to decide what policy is + appropriate. Changes to existing policies can always be initiated + through the normal IETF consensus process, or through the IESG when + appropriate. + + All future RFCs that either explicitly or implicitly rely on IANA to + register or otherwise administer namespace assignments must provide + guidelines for administration of the namespace. + +9.3. After-the-Fact Registrations + + Occasionally, the IETF becomes aware that an unassigned value from a + namespace is in use on the Internet or that an assigned value is + being used for a different purpose than it was registered for. The + IETF does not condone such misuse; procedures of the type described + in this document need to be applied to such cases, and it might not + always be possible to formally assign the desired value. In the + absence of specifications to the contrary, values may only be + reassigned for a different purpose with the consent of the original + assignee (when possible) and with due consideration of the impact of + such a reassignment. In cases of likely controversy, consultation + with the IESG is advised. + + This is part of the reason for the advice in Section 3.1 about using + placeholder values, such as "TBD1", during document development: + problems are often caused by the open use of unregistered values + after results from well-meant, early implementations, where the + implementations retained the use of developmental code points that + never proceeded to a final IANA assignment. + +9.4. Reclaiming Assigned Values + + Reclaiming previously assigned values for reuse is tricky, because + doing so can lead to interoperability problems with deployed systems + still using the assigned values. Moreover, it can be extremely + difficult to determine the extent of deployment of systems making use + of a particular value. However, in cases where the namespace is + running out of unassigned values and additional ones are needed, it + may be desirable to attempt to reclaim unused values. When + reclaiming unused values, the following (at a minimum) should be + considered: + + + + + + +Cotton, et al. Best Current Practice [Page 35] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + o Attempts should be made to contact the original party to which a + value is assigned, to determine if the value was ever used, and if + so, the extent of deployment. (In some cases, products were never + shipped or have long ceased being used. In other cases, it may be + known that a value was never actually used at all.) + + o Reassignments should not normally be made without the concurrence + of the original requester. Reclamation under such conditions + should only take place where there is strong evidence that a value + is not widely used, and the need to reclaim the value outweighs + the cost of a hostile reclamation. IESG Approval is needed in + this case. + + o It may be appropriate to write up the proposed action and solicit + comments from relevant user communities. In some cases, it may be + appropriate to write an RFC that goes through a formal IETF + process (including IETF Last Call) as was done when DHCP reclaimed + some of its "Private Use" options [RFC3942]. + + o It may be useful to differentiate between revocation, release, and + transfer. Revocation occurs when IANA removes an assignment, + release occurs when the assignee initiates that removal, and + transfer occurs when either revocation or release is coupled with + immediate reassignment. It may be useful to specify procedures + for each of these or to explicitly prohibit combinations that are + not desired. + +9.5. Contact Person vs Assignee or Owner + + Many registries include designation of a technical or administrative + contact associated with each entry. Often, this is recorded as + contact information for an individual. It is unclear, though, what + role the individual has with respect to the registration: is this + item registered on behalf of the individual, the company the + individual worked for, or perhaps another organization the individual + was acting for? + + This matters because some time later, when the individual has changed + jobs or roles, and perhaps can no longer be contacted, someone might + want to update the registration. IANA has no way to know what + company, organization, or individual should be allowed to take the + registration over. For registrations rooted in RFCs, the stream + owner (such as the IESG or the IAB) can make an overriding decision. + But in other cases, there is no recourse. + + Registries can include, in addition to a "Contact" field, an + "Assignee" or "Owner" field (also referred to as "Change Controller") + that can be used to address this situation, giving IANA clear + + + +Cotton, et al. Best Current Practice [Page 36] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + guidance as to the actual owner of the registration. This is + strongly advised, especially for registries that do not require RFCs + to manage their information (e.g., registries with policies such as + First Come First Served (Section 4.4), Expert Review (Section 4.5), + and Specification Required (Section 4.6)). Alternatively, + organizations can put an organizational role into the "Contact" field + in order to make their ownership clear. + +9.6. Closing or Obsoleting a Registry/Registrations + + Sometimes there is a request to "close" a registry to further + registrations. When a registry is closed, no further registrations + will be accepted. The information in the registry will still be + valid and registrations already in the registry can still be updated. + + A closed registry can also be marked as "obsolete", as an indication + that the information in the registry is no longer in current use. + + Specific entries in a registry can be marked as "obsolete" (no longer + in use) or "deprecated" (use is not recommended). + + Such changes to registries and registered values are subject to + normal change controls (see Section 2.3). Any closure, obsolescence, + or deprecation serves to annotate the registry involved; the + information in the registry remains there for informational and + historic purposes. + +10. Appeals + + Appeals of protocol parameter registration decisions can be made + using the normal IETF appeals process as described in [RFC2026], + Section 6.5. That is, an initial appeal should be directed to the + IESG, followed (if necessary) by an appeal to the IAB. + +11. Mailing Lists + + All IETF mailing lists associated with evaluating or discussing + assignment requests as described in this document are subject to + whatever rules of conduct and methods of list management are + currently defined by best current practices or by IESG decision. + +12. Security Considerations + + Information that creates or updates a registration needs to be + authenticated and authorized. IANA updates registries according to + instructions in published RFCs and from the IESG. It may also accept + clarifications from document authors, relevant working group chairs, + designated experts, and mail list participants. + + + +Cotton, et al. Best Current Practice [Page 37] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + Information concerning possible security vulnerabilities of a + protocol may change over time. Likewise, security vulnerabilities + related to how an assigned number is used may change as well. As new + vulnerabilities are discovered, information about such + vulnerabilities may need to be attached to existing registrations so + that users are not misled as to the true security issues surrounding + the use of a registered number. + + Security needs to be considered as part of the selection of a + registration policy. For some protocols, registration of certain + parameters will have security implications, and registration policies + for the relevant registries must ensure that requests get appropriate + review with those security implications in mind. + + An analysis of security issues is generally required for all + protocols that make use of parameters (data types, operation codes, + keywords, etc.) documented in IETF protocols or registered by IANA. + Such security considerations are usually included in the protocol + document [BCP72]. It is the responsibility of the IANA + considerations associated with a particular registry to specify + whether value-specific security considerations must be provided when + assigning new values and the process for reviewing such claims. + +13. IANA Considerations + + Sitewide, IANA has replaced references to RFC 5226 with references to + this document. + +14. Changes Relative to Earlier Editions of BCP 26 + +14.1. 2016: Changes in This Document Relative to RFC 5226 + + Significant additions: + + o Removed RFC 2119 key words, boilerplate, and reference, preferring + plain English -- this is not a protocol specification. + + o Added Section 1.1, Keep IANA Considerations for IANA + + o Added Section 1.2, For Updated Information + + o Added Section 2.1, Organization of Registries + + o Added best practice for selecting an appropriate policy into + Section 4. + + o Added Section 4.12, Using Multiple Policies in Combination + + + + +Cotton, et al. Best Current Practice [Page 38] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + o Added Section 2.3, Specifying Change Control for a Registry + + o Added Section 3.4, Early Allocations + + o Moved each well-known policy into a separate subsection of + Section 4. + + o Added Section 5.4, Expert Reviews and the Document Lifecycle + + o Added Section 7, Documentation References in IANA Registries + + o Added Section 8, What to Do in "bis" Documents + + o Added Section 9.5, Contact Person vs Assignee or Owner + + o Added Section 9.6, Closing or Obsoleting a Registry/Registrations + + Clarifications and such: + + o Some reorganization -- moved text around for clarity and easier + reading. + + o Made clarifications about identification of IANA registries and + use of URLs for them. + + o Clarified the distinction between "Unassigned" and "Reserved". + + o Made some clarifications in "Expert Review" about instructions to + the designated expert. + + o Made some clarifications in "Specification Required" about how to + declare this policy. + + o Assorted minor clarifications and editorial changes throughout. + +14.2. 2008: Changes in RFC 5226 Relative to RFC 2434 + + Changes include: + + o Major reordering of text to expand descriptions and to better + group topics such as "updating registries" vs. "creating new + registries", in order to make it easier for authors to find the + text most applicable to their needs. + + o Numerous editorial changes to improve readability. + + + + + + +Cotton, et al. Best Current Practice [Page 39] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + o Changed the term "IETF Consensus" to "IETF Review" and added more + clarifications. History has shown that people see the words "IETF + Consensus" (without consulting the actual definition) and are + quick to make incorrect assumptions about what the term means in + the context of IANA Considerations. + + o Added "RFC Required" to list of defined policies. + + o Much more explicit directions and examples of "what to put in + RFCs". + + o "Specification Required" now implies use of a designated expert to + evaluate specs for sufficient clarity. + + o Added a section describing provisional registrations. + + o Significantly changed the wording in the "Designated Experts" + section. Main purpose is to make clear that Expert Reviewers are + accountable to the community, and to provide some guidance for + review criteria in the default case. + + o Changed wording to remove any special appeals path. The normal + RFC 2026 appeals path is used. + + o Added a section about reclaiming unused values. + + o Added a section on after-the-fact registrations. + + o Added a section indicating that mailing lists used to evaluate + possible assignments (such as by a designated expert) are subject + to normal IETF rules. + +15. References + +15.1. Normative References + + [RFC2026] Bradner, S., "The Internet Standards Process -- Revision + 3", BCP 9, RFC 2026, DOI 10.17487/RFC2026, October 1996, + <http://www.rfc-editor.org/info/rfc2026>. + +15.2. Informative References + + [BCP72] Rescorla, E. and B. Korver, "Guidelines for Writing RFC + Text on Security Considerations", BCP 72, RFC 3552, July + 2003, <http://www.rfc-editor.org/info/bcp72>. + + + + + + +Cotton, et al. Best Current Practice [Page 40] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + [RFC791] Postel, J., "Internet Protocol", STD 5, RFC 791, + DOI 10.17487/RFC0791, September 1981, + <http://www.rfc-editor.org/info/rfc791>. + + [RFC1591] Postel, J., "Domain Name System Structure and Delegation", + RFC 1591, DOI 10.17487/RFC1591, March 1994, + <http://www.rfc-editor.org/info/rfc1591>. + + [RFC2434] Narten, T. and H. Alvestrand, "Guidelines for Writing an + IANA Considerations Section in RFCs", RFC 2434, + DOI 10.17487/RFC2434, October 1998, + <http://www.rfc-editor.org/info/rfc2434>. + + [RFC2860] Carpenter, B., Baker, F., and M. Roberts, "Memorandum of + Understanding Concerning the Technical Work of the + Internet Assigned Numbers Authority", RFC 2860, + DOI 10.17487/RFC2860, June 2000, + <http://www.rfc-editor.org/info/rfc2860>. + + [RFC2939] Droms, R., "Procedures and IANA Guidelines for Definition + of New DHCP Options and Message Types", BCP 43, RFC 2939, + DOI 10.17487/RFC2939, September 2000, + <http://www.rfc-editor.org/info/rfc2939>. + + [RFC3228] Fenner, B., "IANA Considerations for IPv4 Internet Group + Management Protocol (IGMP)", BCP 57, RFC 3228, + DOI 10.17487/RFC3228, February 2002, + <http://www.rfc-editor.org/info/rfc3228>. + + [RFC3575] Aboba, B., "IANA Considerations for RADIUS (Remote + Authentication Dial In User Service)", RFC 3575, + DOI 10.17487/RFC3575, July 2003, + <http://www.rfc-editor.org/info/rfc3575>. + + [RFC3692] Narten, T., "Assigning Experimental and Testing Numbers + Considered Useful", BCP 82, RFC 3692, + DOI 10.17487/RFC3692, January 2004, + <http://www.rfc-editor.org/info/rfc3692>. + + [RFC3748] Aboba, B., Blunk, L., Vollbrecht, J., Carlson, J., and H. + Levkowetz, Ed., "Extensible Authentication Protocol + (EAP)", RFC 3748, DOI 10.17487/RFC3748, June 2004, + <http://www.rfc-editor.org/info/rfc3748>. + + [RFC3864] Klyne, G., Nottingham, M., and J. Mogul, "Registration + Procedures for Message Header Fields", BCP 90, RFC 3864, + DOI 10.17487/RFC3864, September 2004, + <http://www.rfc-editor.org/info/rfc3864>. + + + +Cotton, et al. Best Current Practice [Page 41] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + [RFC3942] Volz, B., "Reclassifying Dynamic Host Configuration + Protocol version 4 (DHCPv4) Options", RFC 3942, + DOI 10.17487/RFC3942, November 2004, + <http://www.rfc-editor.org/info/rfc3942>. + + [RFC3968] Camarillo, G., "The Internet Assigned Number Authority + (IANA) Header Field Parameter Registry for the Session + Initiation Protocol (SIP)", BCP 98, RFC 3968, + DOI 10.17487/RFC3968, December 2004, + <http://www.rfc-editor.org/info/rfc3968>. + + [RFC4025] Richardson, M., "A Method for Storing IPsec Keying + Material in DNS", RFC 4025, DOI 10.17487/RFC4025, March + 2005, <http://www.rfc-editor.org/info/rfc4025>. + + [RFC4044] McCloghrie, K., "Fibre Channel Management MIB", RFC 4044, + DOI 10.17487/RFC4044, May 2005, + <http://www.rfc-editor.org/info/rfc4044>. + + [RFC4124] Le Faucheur, F., Ed., "Protocol Extensions for Support of + Diffserv-aware MPLS Traffic Engineering", RFC 4124, + DOI 10.17487/RFC4124, June 2005, + <http://www.rfc-editor.org/info/rfc4124>. + + [RFC4169] Torvinen, V., Arkko, J., and M. Naslund, "Hypertext + Transfer Protocol (HTTP) Digest Authentication Using + Authentication and Key Agreement (AKA) Version-2", + RFC 4169, DOI 10.17487/RFC4169, November 2005, + <http://www.rfc-editor.org/info/rfc4169>. + + [RFC4271] Rekhter, Y., Ed., Li, T., Ed., and S. Hares, Ed., "A + Border Gateway Protocol 4 (BGP-4)", RFC 4271, + DOI 10.17487/RFC4271, January 2006, + <http://www.rfc-editor.org/info/rfc4271>. + + [RFC4283] Patel, A., Leung, K., Khalil, M., Akhtar, H., and K. + Chowdhury, "Mobile Node Identifier Option for Mobile IPv6 + (MIPv6)", RFC 4283, DOI 10.17487/RFC4283, November 2005, + <http://www.rfc-editor.org/info/rfc4283>. + + [RFC4340] Kohler, E., Handley, M., and S. Floyd, "Datagram + Congestion Control Protocol (DCCP)", RFC 4340, + DOI 10.17487/RFC4340, March 2006, + <http://www.rfc-editor.org/info/rfc4340>. + + + + + + + +Cotton, et al. Best Current Practice [Page 42] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + [RFC4422] Melnikov, A., Ed. and K. Zeilenga, Ed., "Simple + Authentication and Security Layer (SASL)", RFC 4422, + DOI 10.17487/RFC4422, June 2006, + <http://www.rfc-editor.org/info/rfc4422>. + + [RFC4446] Martini, L., "IANA Allocations for Pseudowire Edge to Edge + Emulation (PWE3)", BCP 116, RFC 4446, + DOI 10.17487/RFC4446, April 2006, + <http://www.rfc-editor.org/info/rfc4446>. + + [RFC4520] Zeilenga, K., "Internet Assigned Numbers Authority (IANA) + Considerations for the Lightweight Directory Access + Protocol (LDAP)", BCP 64, RFC 4520, DOI 10.17487/RFC4520, + June 2006, <http://www.rfc-editor.org/info/rfc4520>. + + [RFC4589] Schulzrinne, H. and H. Tschofenig, "Location Types + Registry", RFC 4589, DOI 10.17487/RFC4589, July 2006, + <http://www.rfc-editor.org/info/rfc4589>. + + [RFC4727] Fenner, B., "Experimental Values In IPv4, IPv6, ICMPv4, + ICMPv6, UDP, and TCP Headers", RFC 4727, + DOI 10.17487/RFC4727, November 2006, + <http://www.rfc-editor.org/info/rfc4727>. + + [RFC5246] Dierks, T. and E. Rescorla, "The Transport Layer Security + (TLS) Protocol Version 1.2", RFC 5246, + DOI 10.17487/RFC5246, August 2008, + <http://www.rfc-editor.org/info/rfc5246>. + + [RFC5378] Bradner, S., Ed. and J. Contreras, Ed., "Rights + Contributors Provide to the IETF Trust", BCP 78, RFC 5378, + DOI 10.17487/RFC5378, November 2008, + <http://www.rfc-editor.org/info/rfc5378>. + + [RFC5742] Alvestrand, H. and R. Housley, "IESG Procedures for + Handling of Independent and IRTF Stream Submissions", + BCP 92, RFC 5742, DOI 10.17487/RFC5742, December 2009, + <http://www.rfc-editor.org/info/rfc5742>. + + [RFC5771] Cotton, M., Vegoda, L., and D. Meyer, "IANA Guidelines for + IPv4 Multicast Address Assignments", BCP 51, RFC 5771, + DOI 10.17487/RFC5771, March 2010, + <http://www.rfc-editor.org/info/rfc5771>. + + [RFC5795] Sandlund, K., Pelletier, G., and L-E. Jonsson, "The RObust + Header Compression (ROHC) Framework", RFC 5795, + DOI 10.17487/RFC5795, March 2010, + <http://www.rfc-editor.org/info/rfc5795>. + + + +Cotton, et al. Best Current Practice [Page 43] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + [RFC6014] Hoffman, P., "Cryptographic Algorithm Identifier + Allocation for DNSSEC", RFC 6014, DOI 10.17487/RFC6014, + November 2010, <http://www.rfc-editor.org/info/rfc6014>. + + [RFC6230] Boulton, C., Melanchuk, T., and S. McGlashan, "Media + Control Channel Framework", RFC 6230, + DOI 10.17487/RFC6230, May 2011, + <http://www.rfc-editor.org/info/rfc6230>. + + [RFC6275] Perkins, C., Ed., Johnson, D., and J. Arkko, "Mobility + Support in IPv6", RFC 6275, DOI 10.17487/RFC6275, July + 2011, <http://www.rfc-editor.org/info/rfc6275>. + + [RFC6698] Hoffman, P. and J. Schlyter, "The DNS-Based Authentication + of Named Entities (DANE) Transport Layer Security (TLS) + Protocol: TLSA", RFC 6698, DOI 10.17487/RFC6698, August + 2012, <http://www.rfc-editor.org/info/rfc6698>. + + [RFC6709] Carpenter, B., Aboba, B., Ed., and S. Cheshire, "Design + Considerations for Protocol Extensions", RFC 6709, + DOI 10.17487/RFC6709, September 2012, + <http://www.rfc-editor.org/info/rfc6709>. + + [RFC6838] Freed, N., Klensin, J., and T. Hansen, "Media Type + Specifications and Registration Procedures", BCP 13, + RFC 6838, DOI 10.17487/RFC6838, January 2013, + <http://www.rfc-editor.org/info/rfc6838>. + + [RFC6895] Eastlake 3rd, D., "Domain Name System (DNS) IANA + Considerations", BCP 42, RFC 6895, DOI 10.17487/RFC6895, + April 2013, <http://www.rfc-editor.org/info/rfc6895>. + + [RFC6994] Touch, J., "Shared Use of Experimental TCP Options", + RFC 6994, DOI 10.17487/RFC6994, August 2013, + <http://www.rfc-editor.org/info/rfc6994>. + + [RFC7120] Cotton, M., "Early IANA Allocation of Standards Track Code + Points", BCP 100, RFC 7120, DOI 10.17487/RFC7120, January + 2014, <http://www.rfc-editor.org/info/rfc7120>. + + [RFC7564] Saint-Andre, P. and M. Blanchet, "PRECIS Framework: + Preparation, Enforcement, and Comparison of + Internationalized Strings in Application Protocols", + RFC 7564, DOI 10.17487/RFC7564, May 2015, + <http://www.rfc-editor.org/info/rfc7564>. + + + + + + +Cotton, et al. Best Current Practice [Page 44] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + + [RFC7595] Thaler, D., Ed., Hansen, T., and T. Hardie, "Guidelines + and Registration Procedures for URI Schemes", BCP 35, + RFC 7595, DOI 10.17487/RFC7595, June 2015, + <http://www.rfc-editor.org/info/rfc7595>. + + [RFC7752] Gredler, H., Ed., Medved, J., Previdi, S., Farrel, A., and + S. Ray, "North-Bound Distribution of Link-State and + Traffic Engineering (TE) Information Using BGP", RFC 7752, + DOI 10.17487/RFC7752, March 2016, + <http://www.rfc-editor.org/info/rfc7752>. + + [RFC8141] Saint-Andre, P. and J. Klensin, "Uniform Resource Names + (URNs)", RFC 8141, DOI 10.17487/RFC8141, April 2017, + <http://www.rfc-editor.org/info/rfc8141>. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Cotton, et al. Best Current Practice [Page 45] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + +Acknowledgments for This Document (2017) + + Thomas Narten and Harald Tveit Alvestrand edited the two earlier + editions of this document (RFCs 2434 and 5226), and Thomas continues + his role in this third edition. Much of the text from RFC 5226 + remains in this edition. + + Thank you to Amanda Baber and Pearl Liang for their multiple reviews + and suggestions for making this document as thorough as possible. + + This document has benefited from thorough review and comments by many + people, including Benoit Claise, Alissa Cooper, Adrian Farrel, + Stephen Farrell, Tony Hansen, John Klensin, Kathleen Moriarty, Mark + Nottingham, Pete Resnick, and Joe Touch. + + Special thanks to Mark Nottingham for reorganizing some of the text + for better organization and readability, to Tony Hansen for acting as + document shepherd, and to Brian Haberman and Terry Manderson for + acting as sponsoring ADs. + +Acknowledgments from the Second Edition (2008) + + The original acknowledgments section in RFC 5226 was: + + This document has benefited from specific feedback from Jari Arkko, + Marcelo Bagnulo Braun, Brian Carpenter, Michelle Cotton, Spencer + Dawkins, Barbara Denny, Miguel Garcia, Paul Hoffman, Russ Housley, + John Klensin, Allison Mankin, Blake Ramsdell, Mark Townsley, Magnus + Westerlund, and Bert Wijnen. + +Acknowledgments from the First Edition (1998) + + The original acknowledgments section in RFC 2434 was: + + Jon Postel and Joyce Reynolds provided a detailed explanation on what + IANA needs in order to manage assignments efficiently, and patiently + provided comments on multiple versions of this document. Brian + Carpenter provided helpful comments on earlier versions of the + document. One paragraph in the Security Considerations section was + borrowed from RFC 4288. + + + + + + + + + + + +Cotton, et al. Best Current Practice [Page 46] + +RFC 8126 IANA Considerations Section in RFCs June 2017 + + +Authors' Addresses + + Michelle Cotton + PTI, an affiliate of ICANN + 12025 Waterfront Drive, Suite 300 + Los Angeles, CA 90094-2536 + United States of America + + Phone: +1-424-254-5300 + Email: michelle.cotton@iana.org + URI: https://www.iana.org/ + + + Barry Leiba + Huawei Technologies + + Phone: +1 646 827 0648 + Email: barryleiba@computer.org + URI: http://internetmessagingtechnology.org/ + + + Thomas Narten + IBM Corporation + 3039 Cornwallis Ave., PO Box 12195 - BRQA/502 + Research Triangle Park, NC 27709-2195 + United States of America + + Phone: +1 919 254 7798 + Email: narten@us.ibm.com + + + + + + + + + + + + + + + + + + + + + + +Cotton, et al. Best Current Practice [Page 47] + diff --git a/eval/corpora/rfc/RFC 8174 - Ambiguity of Uppercase vs Lowercase in RFC 2119 Key Words.txt b/eval/corpora/rfc/RFC 8174 - Ambiguity of Uppercase vs Lowercase in RFC 2119 Key Words.txt new file mode 100644 index 00000000..b7f20364 --- /dev/null +++ b/eval/corpora/rfc/RFC 8174 - Ambiguity of Uppercase vs Lowercase in RFC 2119 Key Words.txt @@ -0,0 +1,227 @@ + + + + + + +Internet Engineering Task Force (IETF) B. Leiba +Request for Comments: 8174 Huawei Technologies +BCP: 14 May 2017 +Updates: 2119 +Category: Best Current Practice +ISSN: 2070-1721 + + + Ambiguity of Uppercase vs Lowercase in RFC 2119 Key Words + +Abstract + + RFC 2119 specifies common key words that may be used in protocol + specifications. This document aims to reduce the ambiguity by + clarifying that only UPPERCASE usage of the key words have the + defined special meanings. + +Status of This Memo + + This memo documents an Internet Best Current Practice. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + BCPs is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + http://www.rfc-editor.org/info/rfc8174. + + + + + + + + + + + + + + + + + + + + + +Leiba Best Current Practice [Page 1] + +RFC 8174 RFC 2119 Clarification May 2017 + + +Copyright Notice + + Copyright (c) 2017 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (http://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + +Table of Contents + + 1. Introduction . . . . . . . . . . . . . . . . . . . . . . . . 2 + 2. Clarifying Capitalization of Key Words . . . . . . . . . . . 3 + 3. IANA Considerations . . . . . . . . . . . . . . . . . . . . . 4 + 4. Security Considerations . . . . . . . . . . . . . . . . . . . 4 + 5. Normative References . . . . . . . . . . . . . . . . . . . . 4 + Author's Address . . . . . . . . . . . . . . . . . . . . . . . . 4 + +1. Introduction + + RFC 2119 specifies common key words, such as "MUST", "SHOULD", and + "MAY", that may be used in protocol specifications. It says that the + key words "are often capitalized," which has caused confusion about + how to interpret non-capitalized words such as "must" and "should". + + This document updates RFC 2119 by clarifying that only UPPERCASE + usage of the key words have the defined special meanings. This + document is part of BCP 14. + + + + + + + + + + + + + + + + + +Leiba Best Current Practice [Page 2] + +RFC 8174 RFC 2119 Clarification May 2017 + + +2. Clarifying Capitalization of Key Words + + The following change is made to [RFC2119]: + + === OLD === + In many standards track documents several words are used to signify + the requirements in the specification. These words are often + capitalized. This document defines these words as they should be + interpreted in IETF documents. Authors who follow these guidelines + should incorporate this phrase near the beginning of their document: + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "MAY", and "OPTIONAL" in this + document are to be interpreted as described in RFC 2119. + + + === NEW === + In many IETF documents, several words, when they are in all capitals + as shown below, are used to signify the requirements in the + specification. These capitalized words can bring significant clarity + and consistency to documents because their meanings are well defined. + This document defines how those words are interpreted in IETF + documents when the words are in all capitals. + + o These words can be used as defined here, but using them is not + required. Specifically, normative text does not require the use + of these key words. They are used for clarity and consistency + when that is what's wanted, but a lot of normative text does not + use them and is still normative. + + o The words have the meanings specified herein only when they are in + all capitals. + + o When these words are not capitalized, they have their normal + English meanings and are not affected by this document. + + Authors who follow these guidelines should incorporate this phrase + near the beginning of their document: + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL + NOT", "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", + "MAY", and "OPTIONAL" in this document are to be interpreted as + described in BCP 14 [RFC2119] [RFC8174] when, and only when, they + appear in all capitals, as shown here. + + === END === + + + + + +Leiba Best Current Practice [Page 3] + +RFC 8174 RFC 2119 Clarification May 2017 + + +3. IANA Considerations + + This document does not require any IANA actions. + +4. Security Considerations + + This document is purely procedural; there are no related security + considerations. + +5. Normative References + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <http://www.rfc-editor.org/info/rfc2119>. + +Author's Address + + Barry Leiba + Huawei Technologies + + Phone: +1 646 827 0648 + Email: barryleiba@computer.org + URI: http://internetmessagingtechnology.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + +Leiba Best Current Practice [Page 4] + diff --git a/eval/corpora/rfc/RFC 8336 - The ORIGIN HTTP2 Frame.txt b/eval/corpora/rfc/RFC 8336 - The ORIGIN HTTP2 Frame.txt new file mode 100644 index 00000000..5cf34a28 --- /dev/null +++ b/eval/corpora/rfc/RFC 8336 - The ORIGIN HTTP2 Frame.txt @@ -0,0 +1,619 @@ + + + + + + +Internet Engineering Task Force (IETF) M. Nottingham +Request for Comments: 8336 +Category: Standards Track E. Nygren +ISSN: 2070-1721 Akamai Technologies + March 2018 + + + The ORIGIN HTTP/2 Frame + +Abstract + + This document specifies the ORIGIN frame for HTTP/2, to indicate what + origins are available on a given connection. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc8336. + +Copyright Notice + + Copyright (c) 2018 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + + + + + + + + + +Nottingham & Nygren Standards Track [Page 1] + +RFC 8336 ORIGIN Frames March 2018 + + +Table of Contents + + 1. Introduction . . . . . . . . . . . . . . . . . . . . . . . . 2 + 1.1. Notational Conventions . . . . . . . . . . . . . . . . . 2 + 2. The ORIGIN HTTP/2 Frame . . . . . . . . . . . . . . . . . . . 3 + 2.1. Syntax . . . . . . . . . . . . . . . . . . . . . . . . . 3 + 2.2. Processing ORIGIN Frames . . . . . . . . . . . . . . . . 3 + 2.3. The Origin Set . . . . . . . . . . . . . . . . . . . . . 4 + 2.4. Authority, Push, and Coalescing with ORIGIN . . . . . . . 6 + 3. IANA Considerations . . . . . . . . . . . . . . . . . . . . . 7 + 4. Security Considerations . . . . . . . . . . . . . . . . . . . 7 + 5. References . . . . . . . . . . . . . . . . . . . . . . . . . 8 + 5.1. Normative References . . . . . . . . . . . . . . . . . . 8 + 5.2. Informative References . . . . . . . . . . . . . . . . . 8 + Appendix A. Non-Normative Processing Algorithm . . . . . . . . . 10 + Appendix B. Operational Considerations for Servers . . . . . . . 10 + Authors' Addresses . . . . . . . . . . . . . . . . . . . . . . . 11 + +1. Introduction + + HTTP/2 [RFC7540] allows clients to coalesce different origins + [RFC6454] onto the same connection when certain conditions are met. + However, in some cases, a connection is not usable for a coalesced + origin, so the 421 (Misdirected Request) status code ([RFC7540], + Section 9.1.2) was defined. + + Using a status code in this manner allows clients to recover from + misdirected requests, but at the penalty of adding latency. To + address that, this specification defines a new HTTP/2 frame type, + "ORIGIN", to allow servers to indicate for which origins a connection + is usable. + + Additionally, experience has shown that HTTP/2's requirement to + establish server authority using both DNS and the server's + certificate is onerous. This specification relaxes the requirement + to check DNS when the ORIGIN frame is in use. Doing so has + additional benefits, such as removing the latency associated with + some DNS lookups. + +1.1. Notational Conventions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + + + + +Nottingham & Nygren Standards Track [Page 2] + +RFC 8336 ORIGIN Frames March 2018 + + +2. The ORIGIN HTTP/2 Frame + + This document defines a new HTTP/2 frame type ([RFC7540], Section 4) + called ORIGIN, that allows a server to indicate what origin(s) + [RFC6454] the server would like the client to consider as members of + the Origin Set (Section 2.3) for the connection within which it + occurs. + +2.1. Syntax + + The ORIGIN frame type is 0xc (decimal 12) and contains zero or more + instances of the Origin-Entry field. + + +-------------------------------+-------------------------------+ + | Origin-Entry (*) ... + +-------------------------------+-------------------------------+ + + An Origin-Entry is a length-delimited string: + + +-------------------------------+-------------------------------+ + | Origin-Len (16) | ASCII-Origin? ... + +-------------------------------+-------------------------------+ + + Specifically: + + Origin-Len: An unsigned, 16-bit integer indicating the length, in + octets, of the ASCII-Origin field. + + Origin: An OPTIONAL sequence of characters containing the ASCII + serialization of an origin ([RFC6454], Section 6.2) that the + sender asserts this connection is or could be authoritative for. + + The ORIGIN frame does not define any flags. However, future updates + to this specification MAY define flags. See Section 2.2. + +2.2. Processing ORIGIN Frames + + The ORIGIN frame is a non-critical extension to HTTP/2. Endpoints + that do not support this frame can safely ignore it upon receipt. + + When received by an implementing client, it is used to initialize and + manipulate the Origin Set (see Section 2.3), thereby changing how the + client establishes authority for origin servers (see Section 2.4). + + The ORIGIN frame MUST be sent on stream 0; an ORIGIN frame on any + other stream is invalid and MUST be ignored. + + + + + +Nottingham & Nygren Standards Track [Page 3] + +RFC 8336 ORIGIN Frames March 2018 + + + Likewise, the ORIGIN frame is only valid on connections with the "h2" + protocol identifier or when specifically nominated by the protocol's + definition; it MUST be ignored when received on a connection with the + "h2c" protocol identifier. + + This specification does not define any flags for the ORIGIN frame, + but future updates to this specification (through IETF consensus) + might use them to change its semantics. The first four flags (0x1, + 0x2, 0x4, and 0x8) are reserved for backwards-incompatible changes; + therefore, when any of them are set, the ORIGIN frame containing them + MUST be ignored by clients conforming to this specification, unless + the flag's semantics are understood. The remaining flags are + reserved for backwards-compatible changes and do not affect + processing by clients conformant to this specification. + + The ORIGIN frame describes a property of the connection and therefore + is processed hop by hop. An intermediary MUST NOT forward ORIGIN + frames. Clients configured to use a proxy MUST ignore any ORIGIN + frames received from it. + + Each ASCII-Origin field in the frame's payload MUST be parsed as an + ASCII serialization of an origin ([RFC6454], Section 6.2). If + parsing fails, the field MUST be ignored. + + Note that the ORIGIN frame does not support wildcard names (e.g., + "*.example.com") in Origin-Entry. As a result, sending ORIGIN when a + wildcard certificate is in use effectively disables any origins that + are not explicitly listed in the ORIGIN frame(s) (when the client + understands ORIGIN). + + See Appendix A for an illustrative algorithm for processing ORIGIN + frames. + +2.3. The Origin Set + + The set of origins (as per [RFC6454]) that a given connection might + be used for is known in this specification as the Origin Set. + + By default, the Origin Set for a connection is uninitialized. An + uninitialized Origin Set means that clients apply the coalescing + rules from Section 9.1.1 of [RFC7540]. + + + + + + + + + + +Nottingham & Nygren Standards Track [Page 4] + +RFC 8336 ORIGIN Frames March 2018 + + + When an ORIGIN frame is first received and successfully processed by + a client, the connection's Origin Set is defined to contain an + initial origin. The initial origin is composed from: + + o Scheme: "https" + + o Host: the value sent in Server Name Indication (SNI) ([RFC6066], + Section 3) converted to lower case; if SNI is not present, the + remote address of the connection (i.e., the server's IP address) + + o Port: the remote port of the connection (i.e., the server's port) + + The contents of that ORIGIN frame (and subsequent ones) allow the + server to incrementally add new origins to the Origin Set, as + described in Section 2.2. + + The Origin Set is also affected by the 421 (Misdirected Request) + response status code, as defined in [RFC7540], Section 9.1.2. Upon + receipt of a response with this status code, implementing clients + MUST create the ASCII serialization of the corresponding request's + origin (as per [RFC6454], Section 6.2) and remove it from the + connection's Origin Set, if present. + + Note: When sending an ORIGIN frame to a connection that is + initialized as an alternative service [RFC7838], the initial + Origin Set (Section 2.3) will contain an origin with the + appropriate scheme and hostname (since RFC 7838 specifies that the + origin's hostname be sent in SNI). However, it is possible that + the port will be different than that of the intended origin, since + the initial Origin Set is calculated using the actual port in use, + which can be different for the alternative service. In this case, + the intended origin needs to be sent in the ORIGIN frame + explicitly. + + For example, a client making requests for "https://example.com" is + directed to an alternative service at ("h2", "x.example.net", + "8443"). If this alternative service sends an ORIGIN frame, the + initial origin will be "https://example.com:8443". The client + will not be able to use the alternative service to make requests + for "https://example.com" unless that origin is explicitly + included in the ORIGIN frame. + + + + + + + + + + +Nottingham & Nygren Standards Track [Page 5] + +RFC 8336 ORIGIN Frames March 2018 + + +2.4. Authority, Push, and Coalescing with ORIGIN + + Section 10.1 of [RFC7540] uses both DNS and the presented Transport + Layer Security (TLS) certificate to establish the origin server(s) + that a connection is authoritative for, just as HTTP/1.1 does in + [RFC7230]. + + Furthermore, Section 9.1.1 of [RFC7540] explicitly allows a + connection to be used for more than one origin server, if it is + authoritative. This affects what responses can be considered + authoritative, both for direct responses to requests and for server + push (see [RFC7540], Section 8.2.2). Indirectly, it also affects + what requests will be sent on a connection, since clients will + generally only send requests on connections that they believe to be + authoritative for the origin in question. + + Once an Origin Set has been initialized for a connection, clients + that implement this specification use it to help determine what the + connection is authoritative for. Specifically, such clients MUST NOT + consider a connection to be authoritative for an origin not present + in the Origin Set, and they SHOULD use the connection for all + requests to origins in the Origin Set for which the connection is + authoritative, unless there are operational reasons for opening a new + connection. + + Note that for a connection to be considered authoritative for a given + origin, the server is still required to authenticate with a + certificate that passes suitable checks; see Section 9.1.1 of + [RFC7540] for more information. This includes verifying that the + host matches a "dNSName" value from the certificate "subjectAltName" + field (using the rules defined in [RFC2818]; see also [RFC5280], + Section 4.2.1.6). + + Additionally, clients MAY avoid consulting DNS to establish the + connection's authority for new requests to origins in the Origin Set; + however, those that do so face new risks, as explained in Section 4. + + Because ORIGIN can change the set of origins a connection is used for + over time, it is possible that a client might have more than one + viable connection to an origin open at any time. When this occurs, + clients SHOULD NOT emit new requests on any connection whose Origin + Set is a proper subset of another connection's Origin Set, and they + SHOULD close it once all outstanding requests are satisfied. + + The Origin Set is unaffected by any alternative services [RFC7838] + advertisements made by the server. Advertising an alternative + service does not affect whether a server is authoritative. + + + + +Nottingham & Nygren Standards Track [Page 6] + +RFC 8336 ORIGIN Frames March 2018 + + +3. IANA Considerations + + This specification adds an entry to the "HTTP/2 Frame Type" registry. + + o Frame Type: ORIGIN + + o Code: 0xc + + o Specification: RFC 8336 + +4. Security Considerations + + Clients that blindly trust the ORIGIN frame's contents will be + vulnerable to a large number of attacks. See Section 2.4 for + mitigations. + + Relaxing the requirement to consult DNS when determining authority + for an origin means that an attacker who possesses a valid + certificate no longer needs to be on path to redirect traffic to + them; instead of modifying DNS, they need only convince the user to + visit another website in order to coalesce connections to the target + onto their existing connection. + + As a result, clients opting not to consult DNS ought to employ some + alternative means to establish a high degree of confidence that the + certificate is legitimate. For example, clients might skip + consulting DNS only if they receive proof of inclusion in a + Certificate Transparency log [RFC6962] or if they have a recent + Online Certificate Status Protocol (OCSP) response [RFC6960] + (possibly using the "status_request" TLS extension [RFC6066]) showing + that the certificate was not revoked. + + The Origin Set's size is unbounded by this specification and thus + could be used by attackers to exhaust client resources. To mitigate + this risk, clients can monitor their state commitment and close the + connection if it is too high. + + + + + + + + + + + + + + + +Nottingham & Nygren Standards Track [Page 7] + +RFC 8336 ORIGIN Frames March 2018 + + +5. References + +5.1. Normative References + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC2818] Rescorla, E., "HTTP Over TLS", RFC 2818, + DOI 10.17487/RFC2818, May 2000, + <https://www.rfc-editor.org/info/rfc2818>. + + [RFC5280] Cooper, D., Santesson, S., Farrell, S., Boeyen, S., + Housley, R., and W. Polk, "Internet X.509 Public Key + Infrastructure Certificate and Certificate Revocation List + (CRL) Profile", RFC 5280, DOI 10.17487/RFC5280, May 2008, + <https://www.rfc-editor.org/info/rfc5280>. + + [RFC6066] Eastlake 3rd, D., "Transport Layer Security (TLS) + Extensions: Extension Definitions", RFC 6066, + DOI 10.17487/RFC6066, January 2011, + <https://www.rfc-editor.org/info/rfc6066>. + + [RFC6454] Barth, A., "The Web Origin Concept", RFC 6454, + DOI 10.17487/RFC6454, December 2011, + <https://www.rfc-editor.org/info/rfc6454>. + + [RFC7540] Belshe, M., Peon, R., and M. Thomson, Ed., "Hypertext + Transfer Protocol Version 2 (HTTP/2)", RFC 7540, + DOI 10.17487/RFC7540, May 2015, + <https://www.rfc-editor.org/info/rfc7540>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + +5.2. Informative References + + [RFC6960] Santesson, S., Myers, M., Ankney, R., Malpani, A., + Galperin, S., and C. Adams, "X.509 Internet Public Key + Infrastructure Online Certificate Status Protocol - OCSP", + RFC 6960, DOI 10.17487/RFC6960, June 2013, + <https://www.rfc-editor.org/info/rfc6960>. + + [RFC6962] Laurie, B., Langley, A., and E. Kasper, "Certificate + Transparency", RFC 6962, DOI 10.17487/RFC6962, June 2013, + <https://www.rfc-editor.org/info/rfc6962>. + + + +Nottingham & Nygren Standards Track [Page 8] + +RFC 8336 ORIGIN Frames March 2018 + + + [RFC7230] Fielding, R., Ed. and J. Reschke, Ed., "Hypertext Transfer + Protocol (HTTP/1.1): Message Syntax and Routing", + RFC 7230, DOI 10.17487/RFC7230, June 2014, + <https://www.rfc-editor.org/info/rfc7230>. + + [RFC7838] Nottingham, M., McManus, P., and J. Reschke, "HTTP + Alternative Services", RFC 7838, DOI 10.17487/RFC7838, + April 2016, <https://www.rfc-editor.org/info/rfc7838>. + + [RFC8288] Nottingham, M., "Web Linking", RFC 8288, + DOI 10.17487/RFC8288, October 2017, + <https://www.rfc-editor.org/info/rfc8288>. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Nottingham & Nygren Standards Track [Page 9] + +RFC 8336 ORIGIN Frames March 2018 + + +Appendix A. Non-Normative Processing Algorithm + + The following algorithm illustrates how a client could handle + received ORIGIN frames: + + 1. If the client is configured to use a proxy for the connection, + ignore the frame and stop processing. + + 2. If the connection is not identified with the "h2" protocol + identifier or another protocol that has explicitly opted into + this specification, ignore the frame and stop processing. + + 3. If the frame occurs upon any stream except stream 0, ignore the + frame and stop processing. + + 4. If any of the flags 0x1, 0x2, 0x4, or 0x8 are set, ignore the + frame and stop processing. + + 5. If no previous ORIGIN frame on the connection has reached this + step, initialize the Origin Set as per Section 2.3. + + 6. For each "Origin-Entry" in the frame payload: + + 1. Parse "ASCII-Origin" as an ASCII serialization of an origin + ([RFC6454], Section 6.2), and let the result be + "parsed_origin". If parsing fails, skip to the next + "Origin-Entry". + + 2. Add "parsed_origin" to the Origin Set. + +Appendix B. Operational Considerations for Servers + + The ORIGIN frame allows a server to indicate for which origins a + given connection ought be used. The set of origins advertised using + this mechanism is under control of the server; servers are not + obligated to use it or to advertise all origins that they might be + able to answer a request for. + + For example, it can be used to inform the client that the connection + is to only be used for the SNI-based origin, by sending an empty + ORIGIN frame. Or, a larger number of origins can be indicated by + including a payload. + + Generally, this information is most useful to send before sending any + part of a response that might initiate a new connection; for example, + "Link" response header fields [RFC8288], or links in the response + body. + + + + +Nottingham & Nygren Standards Track [Page 10] + +RFC 8336 ORIGIN Frames March 2018 + + + Therefore, the ORIGIN frame ought be sent as soon as possible on a + connection, ideally before any HEADERS or PUSH_PROMISE frames. + + However, if it's desirable to associate a large number of origins + with a connection, doing so might introduce end-user-perceived + latency, due to their size. As a result, it might be necessary to + select a "core" set of origins to send initially, and expand the set + of origins the connection is used for with subsequent ORIGIN frames + later (e.g., when the connection is idle). + + That said, senders are encouraged to include as many origins as + practical within a single ORIGIN frame; clients need to make + decisions about creating connections on the fly, and if the Origin + Set is split across many frames, their behavior might be suboptimal. + + Senders take note that, as per Section 4, Step 5, of [RFC6454], the + values in an ORIGIN header need to be case-normalized before + serialization. + + Finally, servers that host alternative services [RFC7838] will need + to explicitly advertise their origins when sending ORIGIN, because + the default contents of the Origin Set (as per Section 2.3) do not + contain any alternative services' origins, even if they have been + used previously on the connection. + +Authors' Addresses + + Mark Nottingham + + Email: mnot@mnot.net + URI: https://www.mnot.net/ + + + Erik Nygren + Akamai Technologies + + Email: erik+ietf@nygren.org + + + + + + + + + + + + + + +Nottingham & Nygren Standards Track [Page 11] + diff --git a/eval/corpora/rfc/RFC 8441 - Bootstrapping WebSockets with HTTP2.txt b/eval/corpora/rfc/RFC 8441 - Bootstrapping WebSockets with HTTP2.txt new file mode 100644 index 00000000..e5a5a456 --- /dev/null +++ b/eval/corpora/rfc/RFC 8441 - Bootstrapping WebSockets with HTTP2.txt @@ -0,0 +1,451 @@ + + + + + + +Internet Engineering Task Force (IETF) P. McManus +Request for Comments: 8441 Mozilla +Updates: 6455 September 2018 +Category: Standards Track +ISSN: 2070-1721 + + + Bootstrapping WebSockets with HTTP/2 + +Abstract + + This document defines a mechanism for running the WebSocket Protocol + (RFC 6455) over a single stream of an HTTP/2 connection. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc8441. + +Copyright Notice + + Copyright (c) 2018 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + + + + + + + + + +McManus Standards Track [Page 1] + +RFC 8441 H2 WebSockets September 2018 + + +Table of Contents + + 1. Introduction . . . . . . . . . . . . . . . . . . . . . . . . 2 + 2. Terminology . . . . . . . . . . . . . . . . . . . . . . . . . 3 + 3. The SETTINGS_ENABLE_CONNECT_PROTOCOL SETTINGS Parameter . . . 3 + 4. The Extended CONNECT Method . . . . . . . . . . . . . . . . . 4 + 5. Using Extended CONNECT to Bootstrap the WebSocket Protocol . 4 + 5.1. Example . . . . . . . . . . . . . . . . . . . . . . . . . 6 + 6. Design Considerations . . . . . . . . . . . . . . . . . . . . 6 + 7. About Intermediaries . . . . . . . . . . . . . . . . . . . . 6 + 8. Security Considerations . . . . . . . . . . . . . . . . . . . 7 + 9. IANA Considerations . . . . . . . . . . . . . . . . . . . . . 7 + 9.1. A New HTTP/2 Setting . . . . . . . . . . . . . . . . . . 7 + 9.2. A New HTTP Upgrade Token . . . . . . . . . . . . . . . . 7 + 10. Normative References . . . . . . . . . . . . . . . . . . . . 8 + Acknowledgments . . . . . . . . . . . . . . . . . . . . . . . . . 8 + Author's Address . . . . . . . . . . . . . . . . . . . . . . . . 8 + +1. Introduction + + The Hypertext Transfer Protocol (HTTP) [RFC7230] provides compatible + resource-level semantics across different versions, but it does not + offer compatibility at the connection-management level. Other + protocols that rely on connection-management details of HTTP, such as + WebSockets, must be updated for new versions of HTTP. + + The WebSocket Protocol [RFC6455] uses the HTTP/1.1 Upgrade mechanism + (Section 6.7 of [RFC7230]) to transition a TCP connection from HTTP + into a WebSocket connection. A different approach must be taken with + HTTP/2 [RFC7540]. Due to its multiplexing nature, HTTP/2 does not + allow connection-wide header fields or status codes, such as the + Upgrade and Connection request-header fields or the 101 (Switching + Protocols) response code. These are all required by the [RFC6455] + opening handshake. + + Being able to bootstrap WebSockets from HTTP/2 allows one TCP + connection to be shared by both protocols and extends HTTP/2's more + efficient use of the network to WebSockets. + + This document extends the HTTP CONNECT method, as specified for + HTTP/2 in Section 8.3 of [RFC7540]. The extension allows the + substitution of a new protocol name to connect to rather than the + external host normally used by CONNECT. The result is a tunnel on a + single HTTP/2 stream that can carry data for WebSockets (or any other + protocol). The other streams on the connection may carry more + extended CONNECT tunnels, traditional HTTP/2 data, or a mixture of + both. + + + + +McManus Standards Track [Page 2] + +RFC 8441 H2 WebSockets September 2018 + + + This tunneled stream will be multiplexed with other regular streams + on the connection and enjoys the normal priority, cancellation, and + flow-control features of HTTP/2. + + Streams that successfully establish a WebSocket connection using a + tunneled stream and the modifications to the opening handshake + defined in this document then use the traditional WebSocket Protocol, + treating the stream as if it were the TCP connection in that + specification. + +2. Terminology + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + +3. The SETTINGS_ENABLE_CONNECT_PROTOCOL SETTINGS Parameter + + This document adds a new SETTINGS parameter to those defined by + [RFC7540], Section 6.5.2. + + The new parameter name is SETTINGS_ENABLE_CONNECT_PROTOCOL. The + value of the parameter MUST be 0 or 1. + + Upon receipt of SETTINGS_ENABLE_CONNECT_PROTOCOL with a value of 1, a + client MAY use the Extended CONNECT as defined in this document when + creating new streams. Receipt of this parameter by a server does not + have any impact. + + A sender MUST NOT send a SETTINGS_ENABLE_CONNECT_PROTOCOL parameter + with the value of 0 after previously sending a value of 1. + + Using a SETTINGS parameter to opt into an otherwise incompatible + protocol change is a use of "Extending HTTP/2" defined by Section 5.5 + of [RFC7540]. Specifically, the addition a new pseudo-header field, + ":protocol", and the change in meaning of the :authority pseudo- + header field in Section 4 require opt-in negotiation. If a client + were to use the provisions of the extended CONNECT method defined in + this document without first receiving a + SETTINGS_ENABLE_CONNECT_PROTOCOL parameter, a non-supporting peer + would detect a malformed request and generate a stream error + (Section 8.1.2.6 of [RFC7540]). + + + + + + + +McManus Standards Track [Page 3] + +RFC 8441 H2 WebSockets September 2018 + + +4. The Extended CONNECT Method + + Usage of the CONNECT method in HTTP/2 is defined by Section 8.3 of + [RFC7540]. This extension modifies the method in the following ways: + + o A new pseudo-header field :protocol MAY be included on request + HEADERS indicating the desired protocol to be spoken on the tunnel + created by CONNECT. The pseudo-header field is single valued and + contains a value from the "Hypertext Transfer Protocol (HTTP) + Upgrade Token Registry" located at + <https://www.iana.org/assignments/http-upgrade-tokens/> + + o On requests that contain the :protocol pseudo-header field, the + :scheme and :path pseudo-header fields of the target URI (see + Section 5) MUST also be included. + + o On requests bearing the :protocol pseudo-header field, the + :authority pseudo-header field is interpreted according to + Section 8.1.2.3 of [RFC7540] instead of Section 8.3 of that + document. In particular, the server MUST NOT create a tunnel to + the host indicated by the :authority as it would with a CONNECT + method request that was not modified by this extension. + + Upon receiving a CONNECT request bearing the :protocol pseudo-header + field, the server establishes a tunnel to another service of the + protocol type indicated by the pseudo-header field. This service may + or may not be co-located with the server. + +5. Using Extended CONNECT to Bootstrap the WebSocket Protocol + + The :protocol pseudo-header field MUST be included in the CONNECT + request, and it MUST have a value of "websocket" to initiate a + WebSocket connection on an HTTP/2 stream. Other HTTP request and + response-header fields, such as those for manipulating cookies, may + be included in the HEADERS with the CONNECT method as usual. This + request replaces the GET-based request in [RFC6455] and is used to + process the WebSockets opening handshake. + + The scheme of the target URI (Section 5.1 of [RFC7230]) MUST be + "https" for "wss"-schemed WebSockets and "http" for "ws"-schemed + WebSockets. The remainder of the target URI is the same as the + WebSocket URI. The WebSocket URI is still used for proxy + autoconfiguration. The security requirements for the HTTP/2 + connection used by this specification are established by [RFC7540] + for https requests and [RFC8164] for http requests. + + + + + + +McManus Standards Track [Page 4] + +RFC 8441 H2 WebSockets September 2018 + + + [RFC6455] requires the use of Connection and Upgrade header fields + that are not part of HTTP/2. They MUST NOT be included in the + CONNECT request defined here. + + [RFC6455] requires the use of a Host header field that is also not + part of HTTP/2. The Host information is conveyed as part of the + :authority pseudo-header field, which is required on every HTTP/2 + transaction. + + Implementations using this extended CONNECT to bootstrap WebSockets + do not do the processing of the Sec-WebSocket-Key and Sec-WebSocket- + Accept header fields of [RFC6455] as that functionality has been + superseded by the :protocol pseudo-header field. + + The Origin [RFC6454], Sec-WebSocket-Version, Sec-WebSocket-Protocol, + and Sec-WebSocket-Extensions header fields are used in the CONNECT + request and response-header fields as defined in [RFC6455]. Note + that HTTP/1 header field names were case insensitive, whereas HTTP/2 + requires they be encoded as lowercase. + + After successfully processing the opening handshake, the peers should + proceed with the WebSocket Protocol [RFC6455] using the HTTP/2 stream + from the CONNECT transaction as if it were the TCP connection + referred to in [RFC6455]. The state of the WebSocket connection at + this point is OPEN, as defined by [RFC6455], Section 4.1. + + The HTTP/2 stream closure is also analogous to the TCP connection + closure of [RFC6455]. Orderly TCP-level closures are represented as + END_STREAM flags ([RFC7540], Section 6.1). RST exceptions are + represented with the RST_STREAM frame ([RFC7540], Section 6.4) with + the CANCEL error code ([RFC7540], Section 7). + + + + + + + + + + + + + + + + + + + + +McManus Standards Track [Page 5] + +RFC 8441 H2 WebSockets September 2018 + + +5.1. Example + +[[ From Client ]] [[ From Server ]] + + SETTINGS + SETTINGS_ENABLE_CONNECT_[..] = 1 + +HEADERS + END_HEADERS +:method = CONNECT +:protocol = websocket +:scheme = https +:path = /chat +:authority = server.example.com +sec-websocket-protocol = chat, superchat +sec-websocket-extensions = permessage-deflate +sec-websocket-version = 13 +origin = http://www.example.com + + HEADERS + END_HEADERS + :status = 200 + sec-websocket-protocol = chat + +DATA +WebSocket Data + + DATA + END_STREAM + WebSocket Data + +DATA + END_STREAM +WebSocket Data + +6. Design Considerations + + A more native integration with HTTP/2 is certainly possible with + larger additions to HTTP/2. This design was selected to minimize the + solution complexity while still addressing the primary concern of + running HTTP/2 and WebSockets concurrently. + +7. About Intermediaries + + This document does not change how WebSockets interacts with HTTP + forward proxies. If a client wishing to speak WebSockets connects + via HTTP/2 to an HTTP proxy, it should continue to use a traditional + CONNECT (i.e., not with a :protocol pseudo-header field) to tunnel + through that proxy to the WebSocket server via HTTP. + + + + + + +McManus Standards Track [Page 6] + +RFC 8441 H2 WebSockets September 2018 + + + The resulting version of HTTP on that tunnel determines whether + WebSockets is initiated directly or via a modified CONNECT request + described in this document. + +8. Security Considerations + + [RFC6455] ensures that non-WebSockets clients, especially + XMLHttpRequest-based clients, cannot make a WebSocket connection. + Its primary mechanism for doing that is the use of Sec-prefixed + request-header fields that cannot be created by XMLHttpRequest-based + clients. This specification addresses that concern in two ways: + + o XMLHttpRequest also prohibits use of the CONNECT method in + addition to Sec-prefixed request-header fields. + + o The use of a pseudo-header field is something that is connection + specific, and HTTP/2 never allows a pseudo-header to be created + outside of the protocol stack. + + The security considerations of [RFC6455], Section 10 continue to + apply to the use of the WebSocket Protocol when using this + specification, with the exception of 10.8. That section is not + relevant, because it is specific to the bootstrapping handshake that + is changed in this document. + +9. IANA Considerations + +9.1. A New HTTP/2 Setting + + This document registers an entry in the "HTTP/2 Settings" registry + that was established by Section 11.3 of [RFC7540]. + + Code: 0x8 + Name: SETTINGS_ENABLE_CONNECT_PROTOCOL + Initial Value: 0 + Specification: This document + +9.2. A New HTTP Upgrade Token + + This document registers an entry in the "HTTP Upgrade Tokens" + registry that was established by [RFC7230]. + + Value: websocket + Description: The Web Socket Protocol + Expected Version Tokens: + References: [RFC6455] [RFC8441] + + + + + +McManus Standards Track [Page 7] + +RFC 8441 H2 WebSockets September 2018 + + +10. Normative References + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC6454] Barth, A., "The Web Origin Concept", RFC 6454, + DOI 10.17487/RFC6454, December 2011, + <https://www.rfc-editor.org/info/rfc6454>. + + [RFC6455] Fette, I. and A. Melnikov, "The WebSocket Protocol", + RFC 6455, DOI 10.17487/RFC6455, December 2011, + <https://www.rfc-editor.org/info/rfc6455>. + + [RFC7230] Fielding, R., Ed. and J. Reschke, Ed., "Hypertext Transfer + Protocol (HTTP/1.1): Message Syntax and Routing", + RFC 7230, DOI 10.17487/RFC7230, June 2014, + <https://www.rfc-editor.org/info/rfc7230>. + + [RFC7540] Belshe, M., Peon, R., and M. Thomson, Ed., "Hypertext + Transfer Protocol Version 2 (HTTP/2)", RFC 7540, + DOI 10.17487/RFC7540, May 2015, + <https://www.rfc-editor.org/info/rfc7540>. + + [RFC8164] Nottingham, M. and M. Thomson, "Opportunistic Security for + HTTP/2", RFC 8164, DOI 10.17487/RFC8164, May 2017, + <https://www.rfc-editor.org/info/rfc8164>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + +Acknowledgments + + The 2017 HTTP Workshop had a very productive discussion that helped + determine the key problem and acceptable level of solution + complexity. + +Author's Address + + Patrick McManus + Mozilla + + Email: mcmanus@ducksong.com + + + + + + +McManus Standards Track [Page 8] + diff --git a/eval/corpora/rfc/RFC 8999 - Version-Independent Properties of QUIC.txt b/eval/corpora/rfc/RFC 8999 - Version-Independent Properties of QUIC.txt new file mode 100644 index 00000000..7e0fab24 --- /dev/null +++ b/eval/corpora/rfc/RFC 8999 - Version-Independent Properties of QUIC.txt @@ -0,0 +1,441 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) M. Thomson +Request for Comments: 8999 Mozilla +Category: Standards Track May 2021 +ISSN: 2070-1721 + + + Version-Independent Properties of QUIC + +Abstract + + This document defines the properties of the QUIC transport protocol + that are common to all versions of the protocol. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc8999. + +Copyright Notice + + Copyright (c) 2021 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + +Table of Contents + + 1. An Extremely Abstract Description of QUIC + 2. Fixed Properties of All QUIC Versions + 3. Conventions and Definitions + 4. Notational Conventions + 5. QUIC Packets + 5.1. Long Header + 5.2. Short Header + 5.3. Connection ID + 5.4. Version + 6. Version Negotiation + 7. Security and Privacy Considerations + 8. References + 8.1. Normative References + 8.2. Informative References + Appendix A. Incorrect Assumptions + Author's Address + +1. An Extremely Abstract Description of QUIC + + QUIC is a connection-oriented protocol between two endpoints. Those + endpoints exchange UDP datagrams. These UDP datagrams contain QUIC + packets. QUIC endpoints use QUIC packets to establish a QUIC + connection, which is shared protocol state between those endpoints. + +2. Fixed Properties of All QUIC Versions + + In addition to providing secure, multiplexed transport, QUIC + [QUIC-TRANSPORT] allows for the option to negotiate a version. This + allows the protocol to change over time in response to new + requirements. Many characteristics of the protocol could change + between versions. + + This document describes the subset of QUIC that is intended to remain + stable as new versions are developed and deployed. All of these + invariants are independent of the IP version. + + The primary goal of this document is to ensure that it is possible to + deploy new versions of QUIC. By documenting the properties that + cannot change, this document aims to preserve the ability for QUIC + endpoints to negotiate changes to any other aspect of the protocol. + As a consequence, this also guarantees a minimal amount of + information that is made available to entities other than endpoints. + Unless specifically prohibited in this document, any aspect of the + protocol can change between different versions. + + Appendix A contains a non-exhaustive list of some incorrect + assumptions that might be made based on knowledge of QUIC version 1; + these do not apply to every version of QUIC. + +3. Conventions and Definitions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in BCP + 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + This document defines requirements on future QUIC versions, even + where normative language is not used. + + This document uses terms and notational conventions from + [QUIC-TRANSPORT]. + +4. Notational Conventions + + The format of packets is described using the notation defined in this + section. This notation is the same as that used in [QUIC-TRANSPORT]. + + Complex fields are named and then followed by a list of fields + surrounded by a pair of matching braces. Each field in this list is + separated by commas. + + Individual fields include length information, plus indications about + fixed value, optionality, or repetitions. Individual fields use the + following notational conventions, with all lengths in bits: + + x (A): Indicates that x is A bits long + + x (A..B): Indicates that x can be any length from A to B; A can be + omitted to indicate a minimum of zero bits, and B can be omitted + to indicate no set upper limit; values in this format always end + on a byte boundary + + x (L) = C: Indicates that x has a fixed value of C; the length of x + is described by L, which can use any of the length forms above + + x (L) ...: Indicates that x is repeated zero or more times and that + each instance has a length of L + + This document uses network byte order (that is, big endian) values. + Fields are placed starting from the high-order bits of each byte. + + Figure 1 shows an example structure: + + Example Structure { + One-bit Field (1), + 7-bit Field with Fixed Value (7) = 61, + Arbitrary-Length Field (..), + Variable-Length Field (8..24), + Repeated Field (8) ..., + } + + Figure 1: Example Format + +5. QUIC Packets + + QUIC endpoints exchange UDP datagrams that contain one or more QUIC + packets. This section describes the invariant characteristics of a + QUIC packet. A version of QUIC could permit multiple QUIC packets in + a single UDP datagram, but the invariant properties only describe the + first packet in a datagram. + + QUIC defines two types of packet headers: long and short. Packets + with a long header are identified by the most significant bit of the + first byte being set; packets with a short header have that bit + cleared. + + QUIC packets might be integrity protected, including the header. + However, QUIC Version Negotiation packets are not integrity + protected; see Section 6. + + Aside from the values described here, the payload of QUIC packets is + version specific and of arbitrary length. + +5.1. Long Header + + Long headers take the form described in Figure 2. + + Long Header Packet { + Header Form (1) = 1, + Version-Specific Bits (7), + Version (32), + Destination Connection ID Length (8), + Destination Connection ID (0..2040), + Source Connection ID Length (8), + Source Connection ID (0..2040), + Version-Specific Data (..), + } + + Figure 2: QUIC Long Header + + A QUIC packet with a long header has the high bit of the first byte + set to 1. All other bits in that byte are version specific. + + The next four bytes include a 32-bit Version field. Versions are + described in Section 5.4. + + The next byte contains the length in bytes of the Destination + Connection ID field that follows it. This length is encoded as an + 8-bit unsigned integer. The Destination Connection ID field follows + the Destination Connection ID Length field and is between 0 and 255 + bytes in length. Connection IDs are described in Section 5.3. + + The next byte contains the length in bytes of the Source Connection + ID field that follows it. This length is encoded as an 8-bit + unsigned integer. The Source Connection ID field follows the Source + Connection ID Length field and is between 0 and 255 bytes in length. + + The remainder of the packet contains version-specific content. + +5.2. Short Header + + Short headers take the form described in Figure 3. + + Short Header Packet { + Header Form (1) = 0, + Version-Specific Bits (7), + Destination Connection ID (..), + Version-Specific Data (..), + } + + Figure 3: QUIC Short Header + + A QUIC packet with a short header has the high bit of the first byte + set to 0. + + A QUIC packet with a short header includes a Destination Connection + ID immediately following the first byte. The short header does not + include the Destination Connection ID Length, Source Connection ID + Length, Source Connection ID, or Version fields. The length of the + Destination Connection ID is not encoded in packets with a short + header and is not constrained by this specification. + + The remainder of the packet has version-specific semantics. + +5.3. Connection ID + + A connection ID is an opaque field of arbitrary length. + + The primary function of a connection ID is to ensure that changes in + addressing at lower protocol layers (UDP, IP, and below) do not cause + packets for a QUIC connection to be delivered to the wrong QUIC + endpoint. The connection ID is used by endpoints and the + intermediaries that support them to ensure that each QUIC packet can + be delivered to the correct instance of an endpoint. At the + endpoint, the connection ID is used to identify the QUIC connection + for which the packet is intended. + + The connection ID is chosen by each endpoint using version-specific + methods. Packets for the same QUIC connection might use different + connection ID values. + +5.4. Version + + The Version field contains a 4-byte identifier. This value can be + used by endpoints to identify a QUIC version. A Version field with a + value of 0x00000000 is reserved for version negotiation; see + Section 6. All other values are potentially valid. + + The properties described in this document apply to all versions of + QUIC. A protocol that does not conform to the properties described + in this document is not QUIC. Future documents might describe + additional properties that apply to a specific QUIC version or to a + range of QUIC versions. + +6. Version Negotiation + + A QUIC endpoint that receives a packet with a long header and a + version it either does not understand or does not support might send + a Version Negotiation packet in response. Packets with a short + header do not trigger version negotiation. + + A Version Negotiation packet sets the high bit of the first byte, and + thus it conforms with the format of a packet with a long header as + defined in Section 5.1. A Version Negotiation packet is identifiable + as such by the Version field, which is set to 0x00000000. + + Version Negotiation Packet { + Header Form (1) = 1, + Unused (7), + Version (32) = 0, + Destination Connection ID Length (8), + Destination Connection ID (0..2040), + Source Connection ID Length (8), + Source Connection ID (0..2040), + Supported Version (32) ..., + } + + Figure 4: Version Negotiation Packet + + Only the most significant bit of the first byte of a Version + Negotiation packet has any defined value. The remaining 7 bits, + labeled "Unused", can be set to any value when sending and MUST be + ignored on receipt. + + After the Source Connection ID field, the Version Negotiation packet + contains a list of Supported Version fields, each identifying a + version that the endpoint sending the packet supports. A Version + Negotiation packet contains no other fields. An endpoint MUST ignore + a packet that contains no Supported Version fields or contains a + truncated Supported Version value. + + Version Negotiation packets do not use integrity or confidentiality + protection. Specific QUIC versions might include protocol elements + that allow endpoints to detect modification or corruption in the set + of supported versions. + + An endpoint MUST include the value from the Source Connection ID + field of the packet it receives in the Destination Connection ID + field. The value for the Source Connection ID field MUST be copied + from the Destination Connection ID field of the received packet, + which is initially randomly selected by a client. Echoing both + connection IDs gives clients some assurance that the server received + the packet and that the Version Negotiation packet was not generated + by an attacker that is unable to observe packets. + + An endpoint that receives a Version Negotiation packet might change + the version that it decides to use for subsequent packets. The + conditions under which an endpoint changes its QUIC version will + depend on the version of QUIC that it chooses. + + See [QUIC-TRANSPORT] for a more thorough description of how an + endpoint that supports QUIC version 1 generates and consumes a + Version Negotiation packet. + +7. Security and Privacy Considerations + + It is possible that middleboxes could observe traits of a specific + version of QUIC and assume that when other versions of QUIC exhibit + similar traits the same underlying semantic is being expressed. + There are potentially many such traits; see Appendix A. Some effort + has been made to either eliminate or obscure some observable traits + in QUIC version 1, but many of these remain. Other QUIC versions + might make different design decisions and so exhibit different + traits. + + The QUIC version number does not appear in all QUIC packets, which + means that reliably extracting information from a flow based on + version-specific traits requires that middleboxes retain state for + every connection ID they see. + + The Version Negotiation packet described in this document is not + integrity protected; it only has modest protection against insertion + by attackers. An endpoint MUST authenticate the semantic content of + a Version Negotiation packet if it attempts a different QUIC version + as a result. + +8. References + +8.1. Normative References + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + +8.2. Informative References + + [QUIC-TLS] Thomson, M., Ed. and S. Turner, Ed., "Using TLS to Secure + QUIC", RFC 9001, DOI 10.17487/RFC9001, May 2021, + <https://www.rfc-editor.org/info/rfc9001>. + + [QUIC-TRANSPORT] + Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [RFC5116] McGrew, D., "An Interface and Algorithms for Authenticated + Encryption", RFC 5116, DOI 10.17487/RFC5116, January 2008, + <https://www.rfc-editor.org/info/rfc5116>. + +Appendix A. Incorrect Assumptions + + There are several traits of QUIC version 1 [QUIC-TRANSPORT] that are + not protected from observation but are nonetheless considered to be + changeable when a new version is deployed. + + This section lists a sampling of incorrect assumptions that might be + made about QUIC based on knowledge of QUIC version 1. Some of these + statements are not even true for QUIC version 1. This is not an + exhaustive list; it is intended to be illustrative only. + + *Any and all of the following statements can be false for a given + QUIC version:* + + * QUIC uses TLS [QUIC-TLS], and some TLS messages are visible on the + wire. + + * QUIC long headers are only exchanged during connection + establishment. + + * Every flow on a given 5-tuple will include a connection + establishment phase. + + * The first packets exchanged on a flow use the long header. + + * The last packet before a long period of quiescence might be + assumed to contain only an acknowledgment. + + * QUIC uses an Authenticated Encryption with Associated Data (AEAD) + function (AEAD_AES_128_GCM; see [RFC5116]) to protect the packets + it exchanges during connection establishment. + + * QUIC packet numbers are encrypted and appear as the first + encrypted bytes. + + * QUIC packet numbers increase by one for every packet sent. + + * QUIC has a minimum size for the first handshake packet sent by a + client. + + * QUIC stipulates that a client speak first. + + * QUIC packets always have the second bit of the first byte (0x40) + set. + + * A QUIC Version Negotiation packet is only sent by a server. + + * A QUIC connection ID changes infrequently. + + * QUIC endpoints change the version they speak if they are sent a + Version Negotiation packet. + + * The Version field in a QUIC long header is the same in both + directions. + + * A QUIC packet with a particular value in the Version field means + that the corresponding version of QUIC is in use. + + * Only one connection at a time is established between any pair of + QUIC endpoints. + +Author's Address + + Martin Thomson + Mozilla + + Email: mt@lowentropy.net diff --git a/eval/corpora/rfc/RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt b/eval/corpora/rfc/RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt new file mode 100644 index 00000000..3ceabcf5 --- /dev/null +++ b/eval/corpora/rfc/RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt @@ -0,0 +1,8485 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) J. Iyengar, Ed. +Request for Comments: 9000 Fastly +Category: Standards Track M. Thomson, Ed. +ISSN: 2070-1721 Mozilla + May 2021 + + + QUIC: A UDP-Based Multiplexed and Secure Transport + +Abstract + + This document defines the core of the QUIC transport protocol. QUIC + provides applications with flow-controlled streams for structured + communication, low-latency connection establishment, and network path + migration. QUIC includes security measures that ensure + confidentiality, integrity, and availability in a range of deployment + circumstances. Accompanying documents describe the integration of + TLS for key negotiation, loss detection, and an exemplary congestion + control algorithm. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9000. + +Copyright Notice + + Copyright (c) 2021 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + +Table of Contents + + 1. Overview + 1.1. Document Structure + 1.2. Terms and Definitions + 1.3. Notational Conventions + 2. Streams + 2.1. Stream Types and Identifiers + 2.2. Sending and Receiving Data + 2.3. Stream Prioritization + 2.4. Operations on Streams + 3. Stream States + 3.1. Sending Stream States + 3.2. Receiving Stream States + 3.3. Permitted Frame Types + 3.4. Bidirectional Stream States + 3.5. Solicited State Transitions + 4. Flow Control + 4.1. Data Flow Control + 4.2. Increasing Flow Control Limits + 4.3. Flow Control Performance + 4.4. Handling Stream Cancellation + 4.5. Stream Final Size + 4.6. Controlling Concurrency + 5. Connections + 5.1. Connection ID + 5.1.1. Issuing Connection IDs + 5.1.2. Consuming and Retiring Connection IDs + 5.2. Matching Packets to Connections + 5.2.1. Client Packet Handling + 5.2.2. Server Packet Handling + 5.2.3. Considerations for Simple Load Balancers + 5.3. Operations on Connections + 6. Version Negotiation + 6.1. Sending Version Negotiation Packets + 6.2. Handling Version Negotiation Packets + 6.3. Using Reserved Versions + 7. Cryptographic and Transport Handshake + 7.1. Example Handshake Flows + 7.2. Negotiating Connection IDs + 7.3. Authenticating Connection IDs + 7.4. Transport Parameters + 7.4.1. Values of Transport Parameters for 0-RTT + 7.4.2. New Transport Parameters + 7.5. Cryptographic Message Buffering + 8. Address Validation + 8.1. Address Validation during Connection Establishment + 8.1.1. Token Construction + 8.1.2. Address Validation Using Retry Packets + 8.1.3. Address Validation for Future Connections + 8.1.4. Address Validation Token Integrity + 8.2. Path Validation + 8.2.1. Initiating Path Validation + 8.2.2. Path Validation Responses + 8.2.3. Successful Path Validation + 8.2.4. Failed Path Validation + 9. Connection Migration + 9.1. Probing a New Path + 9.2. Initiating Connection Migration + 9.3. Responding to Connection Migration + 9.3.1. Peer Address Spoofing + 9.3.2. On-Path Address Spoofing + 9.3.3. Off-Path Packet Forwarding + 9.4. Loss Detection and Congestion Control + 9.5. Privacy Implications of Connection Migration + 9.6. Server's Preferred Address + 9.6.1. Communicating a Preferred Address + 9.6.2. Migration to a Preferred Address + 9.6.3. Interaction of Client Migration and Preferred Address + 9.7. Use of IPv6 Flow Label and Migration + 10. Connection Termination + 10.1. Idle Timeout + 10.1.1. Liveness Testing + 10.1.2. Deferring Idle Timeout + 10.2. Immediate Close + 10.2.1. Closing Connection State + 10.2.2. Draining Connection State + 10.2.3. Immediate Close during the Handshake + 10.3. Stateless Reset + 10.3.1. Detecting a Stateless Reset + 10.3.2. Calculating a Stateless Reset Token + 10.3.3. Looping + 11. Error Handling + 11.1. Connection Errors + 11.2. Stream Errors + 12. Packets and Frames + 12.1. Protected Packets + 12.2. Coalescing Packets + 12.3. Packet Numbers + 12.4. Frames and Frame Types + 12.5. Frames and Number Spaces + 13. Packetization and Reliability + 13.1. Packet Processing + 13.2. Generating Acknowledgments + 13.2.1. Sending ACK Frames + 13.2.2. Acknowledgment Frequency + 13.2.3. Managing ACK Ranges + 13.2.4. Limiting Ranges by Tracking ACK Frames + 13.2.5. Measuring and Reporting Host Delay + 13.2.6. ACK Frames and Packet Protection + 13.2.7. PADDING Frames Consume Congestion Window + 13.3. Retransmission of Information + 13.4. Explicit Congestion Notification + 13.4.1. Reporting ECN Counts + 13.4.2. ECN Validation + 14. Datagram Size + 14.1. Initial Datagram Size + 14.2. Path Maximum Transmission Unit + 14.2.1. Handling of ICMP Messages by PMTUD + 14.3. Datagram Packetization Layer PMTU Discovery + 14.3.1. DPLPMTUD and Initial Connectivity + 14.3.2. Validating the Network Path with DPLPMTUD + 14.3.3. Handling of ICMP Messages by DPLPMTUD + 14.4. Sending QUIC PMTU Probes + 14.4.1. PMTU Probes Containing Source Connection ID + 15. Versions + 16. Variable-Length Integer Encoding + 17. Packet Formats + 17.1. Packet Number Encoding and Decoding + 17.2. Long Header Packets + 17.2.1. Version Negotiation Packet + 17.2.2. Initial Packet + 17.2.3. 0-RTT + 17.2.4. Handshake Packet + 17.2.5. Retry Packet + 17.3. Short Header Packets + 17.3.1. 1-RTT Packet + 17.4. Latency Spin Bit + 18. Transport Parameter Encoding + 18.1. Reserved Transport Parameters + 18.2. Transport Parameter Definitions + 19. Frame Types and Formats + 19.1. PADDING Frames + 19.2. PING Frames + 19.3. ACK Frames + 19.3.1. ACK Ranges + 19.3.2. ECN Counts + 19.4. RESET_STREAM Frames + 19.5. STOP_SENDING Frames + 19.6. CRYPTO Frames + 19.7. NEW_TOKEN Frames + 19.8. STREAM Frames + 19.9. MAX_DATA Frames + 19.10. MAX_STREAM_DATA Frames + 19.11. MAX_STREAMS Frames + 19.12. DATA_BLOCKED Frames + 19.13. STREAM_DATA_BLOCKED Frames + 19.14. STREAMS_BLOCKED Frames + 19.15. NEW_CONNECTION_ID Frames + 19.16. RETIRE_CONNECTION_ID Frames + 19.17. PATH_CHALLENGE Frames + 19.18. PATH_RESPONSE Frames + 19.19. CONNECTION_CLOSE Frames + 19.20. HANDSHAKE_DONE Frames + 19.21. Extension Frames + 20. Error Codes + 20.1. Transport Error Codes + 20.2. Application Protocol Error Codes + 21. Security Considerations + 21.1. Overview of Security Properties + 21.1.1. Handshake + 21.1.2. Protected Packets + 21.1.3. Connection Migration + 21.2. Handshake Denial of Service + 21.3. Amplification Attack + 21.4. Optimistic ACK Attack + 21.5. Request Forgery Attacks + 21.5.1. Control Options for Endpoints + 21.5.2. Request Forgery with Client Initial Packets + 21.5.3. Request Forgery with Preferred Addresses + 21.5.4. Request Forgery with Spoofed Migration + 21.5.5. Request Forgery with Version Negotiation + 21.5.6. Generic Request Forgery Countermeasures + 21.6. Slowloris Attacks + 21.7. Stream Fragmentation and Reassembly Attacks + 21.8. Stream Commitment Attack + 21.9. Peer Denial of Service + 21.10. Explicit Congestion Notification Attacks + 21.11. Stateless Reset Oracle + 21.12. Version Downgrade + 21.13. Targeted Attacks by Routing + 21.14. Traffic Analysis + 22. IANA Considerations + 22.1. Registration Policies for QUIC Registries + 22.1.1. Provisional Registrations + 22.1.2. Selecting Codepoints + 22.1.3. Reclaiming Provisional Codepoints + 22.1.4. Permanent Registrations + 22.2. QUIC Versions Registry + 22.3. QUIC Transport Parameters Registry + 22.4. QUIC Frame Types Registry + 22.5. QUIC Transport Error Codes Registry + 23. References + 23.1. Normative References + 23.2. Informative References + Appendix A. Pseudocode + A.1. Sample Variable-Length Integer Decoding + A.2. Sample Packet Number Encoding Algorithm + A.3. Sample Packet Number Decoding Algorithm + A.4. Sample ECN Validation Algorithm + Contributors + Authors' Addresses + +1. Overview + + QUIC is a secure general-purpose transport protocol. This document + defines version 1 of QUIC, which conforms to the version-independent + properties of QUIC defined in [QUIC-INVARIANTS]. + + QUIC is a connection-oriented protocol that creates a stateful + interaction between a client and server. + + The QUIC handshake combines negotiation of cryptographic and + transport parameters. QUIC integrates the TLS handshake [TLS13], + although using a customized framing for protecting packets. The + integration of TLS and QUIC is described in more detail in + [QUIC-TLS]. The handshake is structured to permit the exchange of + application data as soon as possible. This includes an option for + clients to send data immediately (0-RTT), which requires some form of + prior communication or configuration to enable. + + Endpoints communicate in QUIC by exchanging QUIC packets. Most + packets contain frames, which carry control information and + application data between endpoints. QUIC authenticates the entirety + of each packet and encrypts as much of each packet as is practical. + QUIC packets are carried in UDP datagrams [UDP] to better facilitate + deployment in existing systems and networks. + + Application protocols exchange information over a QUIC connection via + streams, which are ordered sequences of bytes. Two types of streams + can be created: bidirectional streams, which allow both endpoints to + send data; and unidirectional streams, which allow a single endpoint + to send data. A credit-based scheme is used to limit stream creation + and to bound the amount of data that can be sent. + + QUIC provides the necessary feedback to implement reliable delivery + and congestion control. An algorithm for detecting and recovering + from loss of data is described in Section 6 of [QUIC-RECOVERY]. QUIC + depends on congestion control to avoid network congestion. An + exemplary congestion control algorithm is described in Section 7 of + [QUIC-RECOVERY]. + + QUIC connections are not strictly bound to a single network path. + Connection migration uses connection identifiers to allow connections + to transfer to a new network path. Only clients are able to migrate + in this version of QUIC. This design also allows connections to + continue after changes in network topology or address mappings, such + as might be caused by NAT rebinding. + + Once established, multiple options are provided for connection + termination. Applications can manage a graceful shutdown, endpoints + can negotiate a timeout period, errors can cause immediate connection + teardown, and a stateless mechanism provides for termination of + connections after one endpoint has lost state. + +1.1. Document Structure + + This document describes the core QUIC protocol and is structured as + follows: + + * Streams are the basic service abstraction that QUIC provides. + + - Section 2 describes core concepts related to streams, + + - Section 3 provides a reference model for stream states, and + + - Section 4 outlines the operation of flow control. + + * Connections are the context in which QUIC endpoints communicate. + + - Section 5 describes core concepts related to connections, + + - Section 6 describes version negotiation, + + - Section 7 details the process for establishing connections, + + - Section 8 describes address validation and critical denial-of- + service mitigations, + + - Section 9 describes how endpoints migrate a connection to a new + network path, + + - Section 10 lists the options for terminating an open + connection, and + + - Section 11 provides guidance for stream and connection error + handling. + + * Packets and frames are the basic unit used by QUIC to communicate. + + - Section 12 describes concepts related to packets and frames, + + - Section 13 defines models for the transmission, retransmission, + and acknowledgment of data, and + + - Section 14 specifies rules for managing the size of datagrams + carrying QUIC packets. + + * Finally, encoding details of QUIC protocol elements are described + in: + + - Section 15 (versions), + + - Section 16 (integer encoding), + + - Section 17 (packet headers), + + - Section 18 (transport parameters), + + - Section 19 (frames), and + + - Section 20 (errors). + + Accompanying documents describe QUIC's loss detection and congestion + control [QUIC-RECOVERY], and the use of TLS and other cryptographic + mechanisms [QUIC-TLS]. + + This document defines QUIC version 1, which conforms to the protocol + invariants in [QUIC-INVARIANTS]. + + To refer to QUIC version 1, cite this document. References to the + limited set of version-independent properties of QUIC can cite + [QUIC-INVARIANTS]. + +1.2. Terms and Definitions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in BCP + 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + Commonly used terms in this document are described below. + + QUIC: The transport protocol described by this document. QUIC is a + name, not an acronym. + + Endpoint: An entity that can participate in a QUIC connection by + generating, receiving, and processing QUIC packets. There are + only two types of endpoints in QUIC: client and server. + + Client: The endpoint that initiates a QUIC connection. + + Server: The endpoint that accepts a QUIC connection. + + QUIC packet: A complete processable unit of QUIC that can be + encapsulated in a UDP datagram. One or more QUIC packets can be + encapsulated in a single UDP datagram. + + Ack-eliciting packet: A QUIC packet that contains frames other than + ACK, PADDING, and CONNECTION_CLOSE. These cause a recipient to + send an acknowledgment; see Section 13.2.1. + + Frame: A unit of structured protocol information. There are + multiple frame types, each of which carries different information. + Frames are contained in QUIC packets. + + Address: When used without qualification, the tuple of IP version, + IP address, and UDP port number that represents one end of a + network path. + + Connection ID: An identifier that is used to identify a QUIC + connection at an endpoint. Each endpoint selects one or more + connection IDs for its peer to include in packets sent towards the + endpoint. This value is opaque to the peer. + + Stream: A unidirectional or bidirectional channel of ordered bytes + within a QUIC connection. A QUIC connection can carry multiple + simultaneous streams. + + Application: An entity that uses QUIC to send and receive data. + + This document uses the terms "QUIC packets", "UDP datagrams", and "IP + packets" to refer to the units of the respective protocols. That is, + one or more QUIC packets can be encapsulated in a UDP datagram, which + is in turn encapsulated in an IP packet. + +1.3. Notational Conventions + + Packet and frame diagrams in this document use a custom format. The + purpose of this format is to summarize, not define, protocol + elements. Prose defines the complete semantics and details of + structures. + + Complex fields are named and then followed by a list of fields + surrounded by a pair of matching braces. Each field in this list is + separated by commas. + + Individual fields include length information, plus indications about + fixed value, optionality, or repetitions. Individual fields use the + following notational conventions, with all lengths in bits: + + x (A): Indicates that x is A bits long + + x (i): Indicates that x holds an integer value using the variable- + length encoding described in Section 16 + + x (A..B): Indicates that x can be any length from A to B; A can be + omitted to indicate a minimum of zero bits, and B can be omitted + to indicate no set upper limit; values in this format always end + on a byte boundary + + x (L) = C: Indicates that x has a fixed value of C; the length of x + is described by L, which can use any of the length forms above + + x (L) = C..D: Indicates that x has a value in the range from C to D, + inclusive, with the length described by L, as above + + [x (L)]: Indicates that x is optional and has a length of L + + x (L) ...: Indicates that x is repeated zero or more times and that + each instance has a length of L + + This document uses network byte order (that is, big endian) values. + Fields are placed starting from the high-order bits of each byte. + + By convention, individual fields reference a complex field by using + the name of the complex field. + + Figure 1 provides an example: + + Example Structure { + One-bit Field (1), + 7-bit Field with Fixed Value (7) = 61, + Field with Variable-Length Integer (i), + Arbitrary-Length Field (..), + Variable-Length Field (8..24), + Field With Minimum Length (16..), + Field With Maximum Length (..128), + [Optional Field (64)], + Repeated Field (8) ..., + } + + Figure 1: Example Format + + When a single-bit field is referenced in prose, the position of that + field can be clarified by using the value of the byte that carries + the field with the field's value set. For example, the value 0x80 + could be used to refer to the single-bit field in the most + significant bit of the byte, such as One-bit Field in Figure 1. + +2. Streams + + Streams in QUIC provide a lightweight, ordered byte-stream + abstraction to an application. Streams can be unidirectional or + bidirectional. + + Streams can be created by sending data. Other processes associated + with stream management -- ending, canceling, and managing flow + control -- are all designed to impose minimal overheads. For + instance, a single STREAM frame (Section 19.8) can open, carry data + for, and close a stream. Streams can also be long-lived and can last + the entire duration of a connection. + + Streams can be created by either endpoint, can concurrently send data + interleaved with other streams, and can be canceled. QUIC does not + provide any means of ensuring ordering between bytes on different + streams. + + QUIC allows for an arbitrary number of streams to operate + concurrently and for an arbitrary amount of data to be sent on any + stream, subject to flow control constraints and stream limits; see + Section 4. + +2.1. Stream Types and Identifiers + + Streams can be unidirectional or bidirectional. Unidirectional + streams carry data in one direction: from the initiator of the stream + to its peer. Bidirectional streams allow for data to be sent in both + directions. + + Streams are identified within a connection by a numeric value, + referred to as the stream ID. A stream ID is a 62-bit integer (0 to + 2^62-1) that is unique for all streams on a connection. Stream IDs + are encoded as variable-length integers; see Section 16. A QUIC + endpoint MUST NOT reuse a stream ID within a connection. + + The least significant bit (0x01) of the stream ID identifies the + initiator of the stream. Client-initiated streams have even-numbered + stream IDs (with the bit set to 0), and server-initiated streams have + odd-numbered stream IDs (with the bit set to 1). + + The second least significant bit (0x02) of the stream ID + distinguishes between bidirectional streams (with the bit set to 0) + and unidirectional streams (with the bit set to 1). + + The two least significant bits from a stream ID therefore identify a + stream as one of four types, as summarized in Table 1. + + +======+==================================+ + | Bits | Stream Type | + +======+==================================+ + | 0x00 | Client-Initiated, Bidirectional | + +------+----------------------------------+ + | 0x01 | Server-Initiated, Bidirectional | + +------+----------------------------------+ + | 0x02 | Client-Initiated, Unidirectional | + +------+----------------------------------+ + | 0x03 | Server-Initiated, Unidirectional | + +------+----------------------------------+ + + Table 1: Stream ID Types + + The stream space for each type begins at the minimum value (0x00 + through 0x03, respectively); successive streams of each type are + created with numerically increasing stream IDs. A stream ID that is + used out of order results in all streams of that type with lower- + numbered stream IDs also being opened. + +2.2. Sending and Receiving Data + + STREAM frames (Section 19.8) encapsulate data sent by an application. + An endpoint uses the Stream ID and Offset fields in STREAM frames to + place data in order. + + Endpoints MUST be able to deliver stream data to an application as an + ordered byte stream. Delivering an ordered byte stream requires that + an endpoint buffer any data that is received out of order, up to the + advertised flow control limit. + + QUIC makes no specific allowances for delivery of stream data out of + order. However, implementations MAY choose to offer the ability to + deliver data out of order to a receiving application. + + An endpoint could receive data for a stream at the same stream offset + multiple times. Data that has already been received can be + discarded. The data at a given offset MUST NOT change if it is sent + multiple times; an endpoint MAY treat receipt of different data at + the same offset within a stream as a connection error of type + PROTOCOL_VIOLATION. + + Streams are an ordered byte-stream abstraction with no other + structure visible to QUIC. STREAM frame boundaries are not expected + to be preserved when data is transmitted, retransmitted after packet + loss, or delivered to the application at a receiver. + + An endpoint MUST NOT send data on any stream without ensuring that it + is within the flow control limits set by its peer. Flow control is + described in detail in Section 4. + +2.3. Stream Prioritization + + Stream multiplexing can have a significant effect on application + performance if resources allocated to streams are correctly + prioritized. + + QUIC does not provide a mechanism for exchanging prioritization + information. Instead, it relies on receiving priority information + from the application. + + A QUIC implementation SHOULD provide ways in which an application can + indicate the relative priority of streams. An implementation uses + information provided by the application to determine how to allocate + resources to active streams. + +2.4. Operations on Streams + + This document does not define an API for QUIC; it instead defines a + set of functions on streams that application protocols can rely upon. + An application protocol can assume that a QUIC implementation + provides an interface that includes the operations described in this + section. An implementation designed for use with a specific + application protocol might provide only those operations that are + used by that protocol. + + On the sending part of a stream, an application protocol can: + + * write data, understanding when stream flow control credit + (Section 4.1) has successfully been reserved to send the written + data; + + * end the stream (clean termination), resulting in a STREAM frame + (Section 19.8) with the FIN bit set; and + + * reset the stream (abrupt termination), resulting in a RESET_STREAM + frame (Section 19.4) if the stream was not already in a terminal + state. + + On the receiving part of a stream, an application protocol can: + + * read data; and + + * abort reading of the stream and request closure, possibly + resulting in a STOP_SENDING frame (Section 19.5). + + An application protocol can also request to be informed of state + changes on streams, including when the peer has opened or reset a + stream, when a peer aborts reading on a stream, when new data is + available, and when data can or cannot be written to the stream due + to flow control. + +3. Stream States + + This section describes streams in terms of their send or receive + components. Two state machines are described: one for the streams on + which an endpoint transmits data (Section 3.1) and another for + streams on which an endpoint receives data (Section 3.2). + + Unidirectional streams use either the sending or receiving state + machine, depending on the stream type and endpoint role. + Bidirectional streams use both state machines at both endpoints. For + the most part, the use of these state machines is the same whether + the stream is unidirectional or bidirectional. The conditions for + opening a stream are slightly more complex for a bidirectional stream + because the opening of either the send or receive side causes the + stream to open in both directions. + + The state machines shown in this section are largely informative. + This document uses stream states to describe rules for when and how + different types of frames can be sent and the reactions that are + expected when different types of frames are received. Though these + state machines are intended to be useful in implementing QUIC, these + states are not intended to constrain implementations. An + implementation can define a different state machine as long as its + behavior is consistent with an implementation that implements these + states. + + | Note: In some cases, a single event or action can cause a + | transition through multiple states. For instance, sending + | STREAM with a FIN bit set can cause two state transitions for a + | sending stream: from the "Ready" state to the "Send" state, and + | from the "Send" state to the "Data Sent" state. + +3.1. Sending Stream States + + Figure 2 shows the states for the part of a stream that sends data to + a peer. + + o + | Create Stream (Sending) + | Peer Creates Bidirectional Stream + v + +-------+ + | Ready | Send RESET_STREAM + | |-----------------------. + +-------+ | + | | + | Send STREAM / | + | STREAM_DATA_BLOCKED | + v | + +-------+ | + | Send | Send RESET_STREAM | + | |---------------------->| + +-------+ | + | | + | Send STREAM + FIN | + v v + +-------+ +-------+ + | Data | Send RESET_STREAM | Reset | + | Sent |------------------>| Sent | + +-------+ +-------+ + | | + | Recv All ACKs | Recv ACK + v v + +-------+ +-------+ + | Data | | Reset | + | Recvd | | Recvd | + +-------+ +-------+ + + Figure 2: States for Sending Parts of Streams + + The sending part of a stream that the endpoint initiates (types 0 and + 2 for clients, 1 and 3 for servers) is opened by the application. + The "Ready" state represents a newly created stream that is able to + accept data from the application. Stream data might be buffered in + this state in preparation for sending. + + Sending the first STREAM or STREAM_DATA_BLOCKED frame causes a + sending part of a stream to enter the "Send" state. An + implementation might choose to defer allocating a stream ID to a + stream until it sends the first STREAM frame and enters this state, + which can allow for better stream prioritization. + + The sending part of a bidirectional stream initiated by a peer (type + 0 for a server, type 1 for a client) starts in the "Ready" state when + the receiving part is created. + + In the "Send" state, an endpoint transmits -- and retransmits as + necessary -- stream data in STREAM frames. The endpoint respects the + flow control limits set by its peer and continues to accept and + process MAX_STREAM_DATA frames. An endpoint in the "Send" state + generates STREAM_DATA_BLOCKED frames if it is blocked from sending by + stream flow control limits (Section 4.1). + + After the application indicates that all stream data has been sent + and a STREAM frame containing the FIN bit is sent, the sending part + of the stream enters the "Data Sent" state. From this state, the + endpoint only retransmits stream data as necessary. The endpoint + does not need to check flow control limits or send + STREAM_DATA_BLOCKED frames for a stream in this state. + MAX_STREAM_DATA frames might be received until the peer receives the + final stream offset. The endpoint can safely ignore any + MAX_STREAM_DATA frames it receives from its peer for a stream in this + state. + + Once all stream data has been successfully acknowledged, the sending + part of the stream enters the "Data Recvd" state, which is a terminal + state. + + From any state that is one of "Ready", "Send", or "Data Sent", an + application can signal that it wishes to abandon transmission of + stream data. Alternatively, an endpoint might receive a STOP_SENDING + frame from its peer. In either case, the endpoint sends a + RESET_STREAM frame, which causes the stream to enter the "Reset Sent" + state. + + An endpoint MAY send a RESET_STREAM as the first frame that mentions + a stream; this causes the sending part of that stream to open and + then immediately transition to the "Reset Sent" state. + + Once a packet containing a RESET_STREAM has been acknowledged, the + sending part of the stream enters the "Reset Recvd" state, which is a + terminal state. + +3.2. Receiving Stream States + + Figure 3 shows the states for the part of a stream that receives data + from a peer. The states for a receiving part of a stream mirror only + some of the states of the sending part of the stream at the peer. + The receiving part of a stream does not track states on the sending + part that cannot be observed, such as the "Ready" state. Instead, + the receiving part of a stream tracks the delivery of data to the + application, some of which cannot be observed by the sender. + + o + | Recv STREAM / STREAM_DATA_BLOCKED / RESET_STREAM + | Create Bidirectional Stream (Sending) + | Recv MAX_STREAM_DATA / STOP_SENDING (Bidirectional) + | Create Higher-Numbered Stream + v + +-------+ + | Recv | Recv RESET_STREAM + | |-----------------------. + +-------+ | + | | + | Recv STREAM + FIN | + v | + +-------+ | + | Size | Recv RESET_STREAM | + | Known |---------------------->| + +-------+ | + | | + | Recv All Data | + v v + +-------+ Recv RESET_STREAM +-------+ + | Data |--- (optional) --->| Reset | + | Recvd | Recv All Data | Recvd | + +-------+<-- (optional) ----+-------+ + | | + | App Read All Data | App Read Reset + v v + +-------+ +-------+ + | Data | | Reset | + | Read | | Read | + +-------+ +-------+ + + Figure 3: States for Receiving Parts of Streams + + The receiving part of a stream initiated by a peer (types 1 and 3 for + a client, or 0 and 2 for a server) is created when the first STREAM, + STREAM_DATA_BLOCKED, or RESET_STREAM frame is received for that + stream. For bidirectional streams initiated by a peer, receipt of a + MAX_STREAM_DATA or STOP_SENDING frame for the sending part of the + stream also creates the receiving part. The initial state for the + receiving part of a stream is "Recv". + + For a bidirectional stream, the receiving part enters the "Recv" + state when the sending part initiated by the endpoint (type 0 for a + client, type 1 for a server) enters the "Ready" state. + + An endpoint opens a bidirectional stream when a MAX_STREAM_DATA or + STOP_SENDING frame is received from the peer for that stream. + Receiving a MAX_STREAM_DATA frame for an unopened stream indicates + that the remote peer has opened the stream and is providing flow + control credit. Receiving a STOP_SENDING frame for an unopened + stream indicates that the remote peer no longer wishes to receive + data on this stream. Either frame might arrive before a STREAM or + STREAM_DATA_BLOCKED frame if packets are lost or reordered. + + Before a stream is created, all streams of the same type with lower- + numbered stream IDs MUST be created. This ensures that the creation + order for streams is consistent on both endpoints. + + In the "Recv" state, the endpoint receives STREAM and + STREAM_DATA_BLOCKED frames. Incoming data is buffered and can be + reassembled into the correct order for delivery to the application. + As data is consumed by the application and buffer space becomes + available, the endpoint sends MAX_STREAM_DATA frames to allow the + peer to send more data. + + When a STREAM frame with a FIN bit is received, the final size of the + stream is known; see Section 4.5. The receiving part of the stream + then enters the "Size Known" state. In this state, the endpoint no + longer needs to send MAX_STREAM_DATA frames; it only receives any + retransmissions of stream data. + + Once all data for the stream has been received, the receiving part + enters the "Data Recvd" state. This might happen as a result of + receiving the same STREAM frame that causes the transition to "Size + Known". After all data has been received, any STREAM or + STREAM_DATA_BLOCKED frames for the stream can be discarded. + + The "Data Recvd" state persists until stream data has been delivered + to the application. Once stream data has been delivered, the stream + enters the "Data Read" state, which is a terminal state. + + Receiving a RESET_STREAM frame in the "Recv" or "Size Known" state + causes the stream to enter the "Reset Recvd" state. This might cause + the delivery of stream data to the application to be interrupted. + + It is possible that all stream data has already been received when a + RESET_STREAM is received (that is, in the "Data Recvd" state). + Similarly, it is possible for remaining stream data to arrive after + receiving a RESET_STREAM frame (the "Reset Recvd" state). An + implementation is free to manage this situation as it chooses. + + Sending a RESET_STREAM means that an endpoint cannot guarantee + delivery of stream data; however, there is no requirement that stream + data not be delivered if a RESET_STREAM is received. An + implementation MAY interrupt delivery of stream data, discard any + data that was not consumed, and signal the receipt of the + RESET_STREAM. A RESET_STREAM signal might be suppressed or withheld + if stream data is completely received and is buffered to be read by + the application. If the RESET_STREAM is suppressed, the receiving + part of the stream remains in "Data Recvd". + + Once the application receives the signal indicating that the stream + was reset, the receiving part of the stream transitions to the "Reset + Read" state, which is a terminal state. + +3.3. Permitted Frame Types + + The sender of a stream sends just three frame types that affect the + state of a stream at either the sender or the receiver: STREAM + (Section 19.8), STREAM_DATA_BLOCKED (Section 19.13), and RESET_STREAM + (Section 19.4). + + A sender MUST NOT send any of these frames from a terminal state + ("Data Recvd" or "Reset Recvd"). A sender MUST NOT send a STREAM or + STREAM_DATA_BLOCKED frame for a stream in the "Reset Sent" state or + any terminal state -- that is, after sending a RESET_STREAM frame. A + receiver could receive any of these three frames in any state, due to + the possibility of delayed delivery of packets carrying them. + + The receiver of a stream sends MAX_STREAM_DATA frames (Section 19.10) + and STOP_SENDING frames (Section 19.5). + + The receiver only sends MAX_STREAM_DATA frames in the "Recv" state. + A receiver MAY send a STOP_SENDING frame in any state where it has + not received a RESET_STREAM frame -- that is, states other than + "Reset Recvd" or "Reset Read". However, there is little value in + sending a STOP_SENDING frame in the "Data Recvd" state, as all stream + data has been received. A sender could receive either of these two + types of frames in any state as a result of delayed delivery of + packets. + +3.4. Bidirectional Stream States + + A bidirectional stream is composed of sending and receiving parts. + Implementations can represent states of the bidirectional stream as + composites of sending and receiving stream states. The simplest + model presents the stream as "open" when either sending or receiving + parts are in a non-terminal state and "closed" when both sending and + receiving streams are in terminal states. + + Table 2 shows a more complex mapping of bidirectional stream states + that loosely correspond to the stream states defined in HTTP/2 + [HTTP2]. This shows that multiple states on sending or receiving + parts of streams are mapped to the same composite state. Note that + this is just one possibility for such a mapping; this mapping + requires that data be acknowledged before the transition to a + "closed" or "half-closed" state. + + +===================+=======================+=================+ + | Sending Part | Receiving Part | Composite State | + +===================+=======================+=================+ + | No Stream / Ready | No Stream / Recv (*1) | idle | + +-------------------+-----------------------+-----------------+ + | Ready / Send / | Recv / Size Known | open | + | Data Sent | | | + +-------------------+-----------------------+-----------------+ + | Ready / Send / | Data Recvd / Data | half-closed | + | Data Sent | Read | (remote) | + +-------------------+-----------------------+-----------------+ + | Ready / Send / | Reset Recvd / Reset | half-closed | + | Data Sent | Read | (remote) | + +-------------------+-----------------------+-----------------+ + | Data Recvd | Recv / Size Known | half-closed | + | | | (local) | + +-------------------+-----------------------+-----------------+ + | Reset Sent / | Recv / Size Known | half-closed | + | Reset Recvd | | (local) | + +-------------------+-----------------------+-----------------+ + | Reset Sent / | Data Recvd / Data | closed | + | Reset Recvd | Read | | + +-------------------+-----------------------+-----------------+ + | Reset Sent / | Reset Recvd / Reset | closed | + | Reset Recvd | Read | | + +-------------------+-----------------------+-----------------+ + | Data Recvd | Data Recvd / Data | closed | + | | Read | | + +-------------------+-----------------------+-----------------+ + | Data Recvd | Reset Recvd / Reset | closed | + | | Read | | + +-------------------+-----------------------+-----------------+ + + Table 2: Possible Mapping of Stream States to HTTP/2 + + | Note (*1): A stream is considered "idle" if it has not yet been + | created or if the receiving part of the stream is in the "Recv" + | state without yet having received any frames. + +3.5. Solicited State Transitions + + If an application is no longer interested in the data it is receiving + on a stream, it can abort reading the stream and specify an + application error code. + + If the stream is in the "Recv" or "Size Known" state, the transport + SHOULD signal this by sending a STOP_SENDING frame to prompt closure + of the stream in the opposite direction. This typically indicates + that the receiving application is no longer reading data it receives + from the stream, but it is not a guarantee that incoming data will be + ignored. + + STREAM frames received after sending a STOP_SENDING frame are still + counted toward connection and stream flow control, even though these + frames can be discarded upon receipt. + + A STOP_SENDING frame requests that the receiving endpoint send a + RESET_STREAM frame. An endpoint that receives a STOP_SENDING frame + MUST send a RESET_STREAM frame if the stream is in the "Ready" or + "Send" state. If the stream is in the "Data Sent" state, the + endpoint MAY defer sending the RESET_STREAM frame until the packets + containing outstanding data are acknowledged or declared lost. If + any outstanding data is declared lost, the endpoint SHOULD send a + RESET_STREAM frame instead of retransmitting the data. + + An endpoint SHOULD copy the error code from the STOP_SENDING frame to + the RESET_STREAM frame it sends, but it can use any application error + code. An endpoint that sends a STOP_SENDING frame MAY ignore the + error code in any RESET_STREAM frames subsequently received for that + stream. + + STOP_SENDING SHOULD only be sent for a stream that has not been reset + by the peer. STOP_SENDING is most useful for streams in the "Recv" + or "Size Known" state. + + An endpoint is expected to send another STOP_SENDING frame if a + packet containing a previous STOP_SENDING is lost. However, once + either all stream data or a RESET_STREAM frame has been received for + the stream -- that is, the stream is in any state other than "Recv" + or "Size Known" -- sending a STOP_SENDING frame is unnecessary. + + An endpoint that wishes to terminate both directions of a + bidirectional stream can terminate one direction by sending a + RESET_STREAM frame, and it can encourage prompt termination in the + opposite direction by sending a STOP_SENDING frame. + +4. Flow Control + + Receivers need to limit the amount of data that they are required to + buffer, in order to prevent a fast sender from overwhelming them or a + malicious sender from consuming a large amount of memory. To enable + a receiver to limit memory commitments for a connection, streams are + flow controlled both individually and across a connection as a whole. + A QUIC receiver controls the maximum amount of data the sender can + send on a stream as well as across all streams at any time, as + described in Sections 4.1 and 4.2. + + Similarly, to limit concurrency within a connection, a QUIC endpoint + controls the maximum cumulative number of streams that its peer can + initiate, as described in Section 4.6. + + Data sent in CRYPTO frames is not flow controlled in the same way as + stream data. QUIC relies on the cryptographic protocol + implementation to avoid excessive buffering of data; see [QUIC-TLS]. + To avoid excessive buffering at multiple layers, QUIC implementations + SHOULD provide an interface for the cryptographic protocol + implementation to communicate its buffering limits. + +4.1. Data Flow Control + + QUIC employs a limit-based flow control scheme where a receiver + advertises the limit of total bytes it is prepared to receive on a + given stream or for the entire connection. This leads to two levels + of data flow control in QUIC: + + * Stream flow control, which prevents a single stream from consuming + the entire receive buffer for a connection by limiting the amount + of data that can be sent on each stream. + + * Connection flow control, which prevents senders from exceeding a + receiver's buffer capacity for the connection by limiting the + total bytes of stream data sent in STREAM frames on all streams. + + Senders MUST NOT send data in excess of either limit. + + A receiver sets initial limits for all streams through transport + parameters during the handshake (Section 7.4). Subsequently, a + receiver sends MAX_STREAM_DATA frames (Section 19.10) or MAX_DATA + frames (Section 19.9) to the sender to advertise larger limits. + + A receiver can advertise a larger limit for a stream by sending a + MAX_STREAM_DATA frame with the corresponding stream ID. A + MAX_STREAM_DATA frame indicates the maximum absolute byte offset of a + stream. A receiver could determine the flow control offset to be + advertised based on the current offset of data consumed on that + stream. + + A receiver can advertise a larger limit for a connection by sending a + MAX_DATA frame, which indicates the maximum of the sum of the + absolute byte offsets of all streams. A receiver maintains a + cumulative sum of bytes received on all streams, which is used to + check for violations of the advertised connection or stream data + limits. A receiver could determine the maximum data limit to be + advertised based on the sum of bytes consumed on all streams. + + Once a receiver advertises a limit for the connection or a stream, it + is not an error to advertise a smaller limit, but the smaller limit + has no effect. + + A receiver MUST close the connection with an error of type + FLOW_CONTROL_ERROR if the sender violates the advertised connection + or stream data limits; see Section 11 for details on error handling. + + A sender MUST ignore any MAX_STREAM_DATA or MAX_DATA frames that do + not increase flow control limits. + + If a sender has sent data up to the limit, it will be unable to send + new data and is considered blocked. A sender SHOULD send a + STREAM_DATA_BLOCKED or DATA_BLOCKED frame to indicate to the receiver + that it has data to write but is blocked by flow control limits. If + a sender is blocked for a period longer than the idle timeout + (Section 10.1), the receiver might close the connection even when the + sender has data that is available for transmission. To keep the + connection from closing, a sender that is flow control limited SHOULD + periodically send a STREAM_DATA_BLOCKED or DATA_BLOCKED frame when it + has no ack-eliciting packets in flight. + +4.2. Increasing Flow Control Limits + + Implementations decide when and how much credit to advertise in + MAX_STREAM_DATA and MAX_DATA frames, but this section offers a few + considerations. + + To avoid blocking a sender, a receiver MAY send a MAX_STREAM_DATA or + MAX_DATA frame multiple times within a round trip or send it early + enough to allow time for loss of the frame and subsequent recovery. + + Control frames contribute to connection overhead. Therefore, + frequently sending MAX_STREAM_DATA and MAX_DATA frames with small + changes is undesirable. On the other hand, if updates are less + frequent, larger increments to limits are necessary to avoid blocking + a sender, requiring larger resource commitments at the receiver. + There is a trade-off between resource commitment and overhead when + determining how large a limit is advertised. + + A receiver can use an autotuning mechanism to tune the frequency and + amount of advertised additional credit based on a round-trip time + estimate and the rate at which the receiving application consumes + data, similar to common TCP implementations. As an optimization, an + endpoint could send frames related to flow control only when there + are other frames to send, ensuring that flow control does not cause + extra packets to be sent. + + A blocked sender is not required to send STREAM_DATA_BLOCKED or + DATA_BLOCKED frames. Therefore, a receiver MUST NOT wait for a + STREAM_DATA_BLOCKED or DATA_BLOCKED frame before sending a + MAX_STREAM_DATA or MAX_DATA frame; doing so could result in the + sender being blocked for the rest of the connection. Even if the + sender sends these frames, waiting for them will result in the sender + being blocked for at least an entire round trip. + + When a sender receives credit after being blocked, it might be able + to send a large amount of data in response, resulting in short-term + congestion; see Section 7.7 of [QUIC-RECOVERY] for a discussion of + how a sender can avoid this congestion. + +4.3. Flow Control Performance + + If an endpoint cannot ensure that its peer always has available flow + control credit that is greater than the peer's bandwidth-delay + product on this connection, its receive throughput will be limited by + flow control. + + Packet loss can cause gaps in the receive buffer, preventing the + application from consuming data and freeing up receive buffer space. + + Sending timely updates of flow control limits can improve + performance. Sending packets only to provide flow control updates + can increase network load and adversely affect performance. Sending + flow control updates along with other frames, such as ACK frames, + reduces the cost of those updates. + +4.4. Handling Stream Cancellation + + Endpoints need to eventually agree on the amount of flow control + credit that has been consumed on every stream, to be able to account + for all bytes for connection-level flow control. + + On receipt of a RESET_STREAM frame, an endpoint will tear down state + for the matching stream and ignore further data arriving on that + stream. + + RESET_STREAM terminates one direction of a stream abruptly. For a + bidirectional stream, RESET_STREAM has no effect on data flow in the + opposite direction. Both endpoints MUST maintain flow control state + for the stream in the unterminated direction until that direction + enters a terminal state. + +4.5. Stream Final Size + + The final size is the amount of flow control credit that is consumed + by a stream. Assuming that every contiguous byte on the stream was + sent once, the final size is the number of bytes sent. More + generally, this is one higher than the offset of the byte with the + largest offset sent on the stream, or zero if no bytes were sent. + + A sender always communicates the final size of a stream to the + receiver reliably, no matter how the stream is terminated. The final + size is the sum of the Offset and Length fields of a STREAM frame + with a FIN flag, noting that these fields might be implicit. + Alternatively, the Final Size field of a RESET_STREAM frame carries + this value. This guarantees that both endpoints agree on how much + flow control credit was consumed by the sender on that stream. + + An endpoint will know the final size for a stream when the receiving + part of the stream enters the "Size Known" or "Reset Recvd" state + (Section 3). The receiver MUST use the final size of the stream to + account for all bytes sent on the stream in its connection-level flow + controller. + + An endpoint MUST NOT send data on a stream at or beyond the final + size. + + Once a final size for a stream is known, it cannot change. If a + RESET_STREAM or STREAM frame is received indicating a change in the + final size for the stream, an endpoint SHOULD respond with an error + of type FINAL_SIZE_ERROR; see Section 11 for details on error + handling. A receiver SHOULD treat receipt of data at or beyond the + final size as an error of type FINAL_SIZE_ERROR, even after a stream + is closed. Generating these errors is not mandatory, because + requiring that an endpoint generate these errors also means that the + endpoint needs to maintain the final size state for closed streams, + which could mean a significant state commitment. + +4.6. Controlling Concurrency + + An endpoint limits the cumulative number of incoming streams a peer + can open. Only streams with a stream ID less than "(max_streams * 4 + + first_stream_id_of_type)" can be opened; see Table 1. Initial + limits are set in the transport parameters; see Section 18.2. + Subsequent limits are advertised using MAX_STREAMS frames; see + Section 19.11. Separate limits apply to unidirectional and + bidirectional streams. + + If a max_streams transport parameter or a MAX_STREAMS frame is + received with a value greater than 2^60, this would allow a maximum + stream ID that cannot be expressed as a variable-length integer; see + Section 16. If either is received, the connection MUST be closed + immediately with a connection error of type TRANSPORT_PARAMETER_ERROR + if the offending value was received in a transport parameter or of + type FRAME_ENCODING_ERROR if it was received in a frame; see + Section 10.2. + + Endpoints MUST NOT exceed the limit set by their peer. An endpoint + that receives a frame with a stream ID exceeding the limit it has + sent MUST treat this as a connection error of type + STREAM_LIMIT_ERROR; see Section 11 for details on error handling. + + Once a receiver advertises a stream limit using the MAX_STREAMS + frame, advertising a smaller limit has no effect. MAX_STREAMS frames + that do not increase the stream limit MUST be ignored. + + As with stream and connection flow control, this document leaves + implementations to decide when and how many streams should be + advertised to a peer via MAX_STREAMS. Implementations might choose + to increase limits as streams are closed, to keep the number of + streams available to peers roughly consistent. + + An endpoint that is unable to open a new stream due to the peer's + limits SHOULD send a STREAMS_BLOCKED frame (Section 19.14). This + signal is considered useful for debugging. An endpoint MUST NOT wait + to receive this signal before advertising additional credit, since + doing so will mean that the peer will be blocked for at least an + entire round trip, and potentially indefinitely if the peer chooses + not to send STREAMS_BLOCKED frames. + +5. Connections + + A QUIC connection is shared state between a client and a server. + + Each connection starts with a handshake phase, during which the two + endpoints establish a shared secret using the cryptographic handshake + protocol [QUIC-TLS] and negotiate the application protocol. The + handshake (Section 7) confirms that both endpoints are willing to + communicate (Section 8.1) and establishes parameters for the + connection (Section 7.4). + + An application protocol can use the connection during the handshake + phase with some limitations. 0-RTT allows application data to be + sent by a client before receiving a response from the server. + However, 0-RTT provides no protection against replay attacks; see + Section 9.2 of [QUIC-TLS]. A server can also send application data + to a client before it receives the final cryptographic handshake + messages that allow it to confirm the identity and liveness of the + client. These capabilities allow an application protocol to offer + the option of trading some security guarantees for reduced latency. + + The use of connection IDs (Section 5.1) allows connections to migrate + to a new network path, both as a direct choice of an endpoint and + when forced by a change in a middlebox. Section 9 describes + mitigations for the security and privacy issues associated with + migration. + + For connections that are no longer needed or desired, there are + several ways for a client and server to terminate a connection, as + described in Section 10. + +5.1. Connection ID + + Each connection possesses a set of connection identifiers, or + connection IDs, each of which can identify the connection. + Connection IDs are independently selected by endpoints; each endpoint + selects the connection IDs that its peer uses. + + The primary function of a connection ID is to ensure that changes in + addressing at lower protocol layers (UDP, IP) do not cause packets + for a QUIC connection to be delivered to the wrong endpoint. Each + endpoint selects connection IDs using an implementation-specific (and + perhaps deployment-specific) method that will allow packets with that + connection ID to be routed back to the endpoint and to be identified + by the endpoint upon receipt. + + Multiple connection IDs are used so that endpoints can send packets + that cannot be identified by an observer as being for the same + connection without cooperation from an endpoint; see Section 9.5. + + Connection IDs MUST NOT contain any information that can be used by + an external observer (that is, one that does not cooperate with the + issuer) to correlate them with other connection IDs for the same + connection. As a trivial example, this means the same connection ID + MUST NOT be issued more than once on the same connection. + + Packets with long headers include Source Connection ID and + Destination Connection ID fields. These fields are used to set the + connection IDs for new connections; see Section 7.2 for details. + + Packets with short headers (Section 17.3) only include the + Destination Connection ID and omit the explicit length. The length + of the Destination Connection ID field is expected to be known to + endpoints. Endpoints using a load balancer that routes based on + connection ID could agree with the load balancer on a fixed length + for connection IDs or agree on an encoding scheme. A fixed portion + could encode an explicit length, which allows the entire connection + ID to vary in length and still be used by the load balancer. + + A Version Negotiation (Section 17.2.1) packet echoes the connection + IDs selected by the client, both to ensure correct routing toward the + client and to demonstrate that the packet is in response to a packet + sent by the client. + + A zero-length connection ID can be used when a connection ID is not + needed to route to the correct endpoint. However, multiplexing + connections on the same local IP address and port while using zero- + length connection IDs will cause failures in the presence of peer + connection migration, NAT rebinding, and client port reuse. An + endpoint MUST NOT use the same IP address and port for multiple + concurrent connections with zero-length connection IDs, unless it is + certain that those protocol features are not in use. + + When an endpoint uses a non-zero-length connection ID, it needs to + ensure that the peer has a supply of connection IDs from which to + choose for packets sent to the endpoint. These connection IDs are + supplied by the endpoint using the NEW_CONNECTION_ID frame + (Section 19.15). + +5.1.1. Issuing Connection IDs + + Each connection ID has an associated sequence number to assist in + detecting when NEW_CONNECTION_ID or RETIRE_CONNECTION_ID frames refer + to the same value. The initial connection ID issued by an endpoint + is sent in the Source Connection ID field of the long packet header + (Section 17.2) during the handshake. The sequence number of the + initial connection ID is 0. If the preferred_address transport + parameter is sent, the sequence number of the supplied connection ID + is 1. + + Additional connection IDs are communicated to the peer using + NEW_CONNECTION_ID frames (Section 19.15). The sequence number on + each newly issued connection ID MUST increase by 1. The connection + ID that a client selects for the first Destination Connection ID + field it sends and any connection ID provided by a Retry packet are + not assigned sequence numbers. + + When an endpoint issues a connection ID, it MUST accept packets that + carry this connection ID for the duration of the connection or until + its peer invalidates the connection ID via a RETIRE_CONNECTION_ID + frame (Section 19.16). Connection IDs that are issued and not + retired are considered active; any active connection ID is valid for + use with the current connection at any time, in any packet type. + This includes the connection ID issued by the server via the + preferred_address transport parameter. + + An endpoint SHOULD ensure that its peer has a sufficient number of + available and unused connection IDs. Endpoints advertise the number + of active connection IDs they are willing to maintain using the + active_connection_id_limit transport parameter. An endpoint MUST NOT + provide more connection IDs than the peer's limit. An endpoint MAY + send connection IDs that temporarily exceed a peer's limit if the + NEW_CONNECTION_ID frame also requires the retirement of any excess, + by including a sufficiently large value in the Retire Prior To field. + + A NEW_CONNECTION_ID frame might cause an endpoint to add some active + connection IDs and retire others based on the value of the Retire + Prior To field. After processing a NEW_CONNECTION_ID frame and + adding and retiring active connection IDs, if the number of active + connection IDs exceeds the value advertised in its + active_connection_id_limit transport parameter, an endpoint MUST + close the connection with an error of type CONNECTION_ID_LIMIT_ERROR. + + An endpoint SHOULD supply a new connection ID when the peer retires a + connection ID. If an endpoint provided fewer connection IDs than the + peer's active_connection_id_limit, it MAY supply a new connection ID + when it receives a packet with a previously unused connection ID. An + endpoint MAY limit the total number of connection IDs issued for each + connection to avoid the risk of running out of connection IDs; see + Section 10.3.2. An endpoint MAY also limit the issuance of + connection IDs to reduce the amount of per-path state it maintains, + such as path validation status, as its peer might interact with it + over as many paths as there are issued connection IDs. + + An endpoint that initiates migration and requires non-zero-length + connection IDs SHOULD ensure that the pool of connection IDs + available to its peer allows the peer to use a new connection ID on + migration, as the peer will be unable to respond if the pool is + exhausted. + + An endpoint that selects a zero-length connection ID during the + handshake cannot issue a new connection ID. A zero-length + Destination Connection ID field is used in all packets sent toward + such an endpoint over any network path. + +5.1.2. Consuming and Retiring Connection IDs + + An endpoint can change the connection ID it uses for a peer to + another available one at any time during the connection. An endpoint + consumes connection IDs in response to a migrating peer; see + Section 9.5 for more details. + + An endpoint maintains a set of connection IDs received from its peer, + any of which it can use when sending packets. When the endpoint + wishes to remove a connection ID from use, it sends a + RETIRE_CONNECTION_ID frame to its peer. Sending a + RETIRE_CONNECTION_ID frame indicates that the connection ID will not + be used again and requests that the peer replace it with a new + connection ID using a NEW_CONNECTION_ID frame. + + As discussed in Section 9.5, endpoints limit the use of a connection + ID to packets sent from a single local address to a single + destination address. Endpoints SHOULD retire connection IDs when + they are no longer actively using either the local or destination + address for which the connection ID was used. + + An endpoint might need to stop accepting previously issued connection + IDs in certain circumstances. Such an endpoint can cause its peer to + retire connection IDs by sending a NEW_CONNECTION_ID frame with an + increased Retire Prior To field. The endpoint SHOULD continue to + accept the previously issued connection IDs until they are retired by + the peer. If the endpoint can no longer process the indicated + connection IDs, it MAY close the connection. + + Upon receipt of an increased Retire Prior To field, the peer MUST + stop using the corresponding connection IDs and retire them with + RETIRE_CONNECTION_ID frames before adding the newly provided + connection ID to the set of active connection IDs. This ordering + allows an endpoint to replace all active connection IDs without the + possibility of a peer having no available connection IDs and without + exceeding the limit the peer sets in the active_connection_id_limit + transport parameter; see Section 18.2. Failure to cease using the + connection IDs when requested can result in connection failures, as + the issuing endpoint might be unable to continue using the connection + IDs with the active connection. + + An endpoint SHOULD limit the number of connection IDs it has retired + locally for which RETIRE_CONNECTION_ID frames have not yet been + acknowledged. An endpoint SHOULD allow for sending and tracking a + number of RETIRE_CONNECTION_ID frames of at least twice the value of + the active_connection_id_limit transport parameter. An endpoint MUST + NOT forget a connection ID without retiring it, though it MAY choose + to treat having connection IDs in need of retirement that exceed this + limit as a connection error of type CONNECTION_ID_LIMIT_ERROR. + + Endpoints SHOULD NOT issue updates of the Retire Prior To field + before receiving RETIRE_CONNECTION_ID frames that retire all + connection IDs indicated by the previous Retire Prior To value. + +5.2. Matching Packets to Connections + + Incoming packets are classified on receipt. Packets can either be + associated with an existing connection or -- for servers -- + potentially create a new connection. + + Endpoints try to associate a packet with an existing connection. If + the packet has a non-zero-length Destination Connection ID + corresponding to an existing connection, QUIC processes that packet + accordingly. Note that more than one connection ID can be associated + with a connection; see Section 5.1. + + If the Destination Connection ID is zero length and the addressing + information in the packet matches the addressing information the + endpoint uses to identify a connection with a zero-length connection + ID, QUIC processes the packet as part of that connection. An + endpoint can use just destination IP and port or both source and + destination addresses for identification, though this makes + connections fragile as described in Section 5.1. + + Endpoints can send a Stateless Reset (Section 10.3) for any packets + that cannot be attributed to an existing connection. A Stateless + Reset allows a peer to more quickly identify when a connection + becomes unusable. + + Packets that are matched to an existing connection are discarded if + the packets are inconsistent with the state of that connection. For + example, packets are discarded if they indicate a different protocol + version than that of the connection or if the removal of packet + protection is unsuccessful once the expected keys are available. + + Invalid packets that lack strong integrity protection, such as + Initial, Retry, or Version Negotiation, MAY be discarded. An + endpoint MUST generate a connection error if processing the contents + of these packets prior to discovering an error, or fully revert any + changes made during that processing. + +5.2.1. Client Packet Handling + + Valid packets sent to clients always include a Destination Connection + ID that matches a value the client selects. Clients that choose to + receive zero-length connection IDs can use the local address and port + to identify a connection. Packets that do not match an existing + connection -- based on Destination Connection ID or, if this value is + zero length, local IP address and port -- are discarded. + + Due to packet reordering or loss, a client might receive packets for + a connection that are encrypted with a key it has not yet computed. + The client MAY drop these packets, or it MAY buffer them in + anticipation of later packets that allow it to compute the key. + + If a client receives a packet that uses a different version than it + initially selected, it MUST discard that packet. + +5.2.2. Server Packet Handling + + If a server receives a packet that indicates an unsupported version + and if the packet is large enough to initiate a new connection for + any supported version, the server SHOULD send a Version Negotiation + packet as described in Section 6.1. A server MAY limit the number of + packets to which it responds with a Version Negotiation packet. + Servers MUST drop smaller packets that specify unsupported versions. + + The first packet for an unsupported version can use different + semantics and encodings for any version-specific field. In + particular, different packet protection keys might be used for + different versions. Servers that do not support a particular version + are unlikely to be able to decrypt the payload of the packet or + properly interpret the result. Servers SHOULD respond with a Version + Negotiation packet, provided that the datagram is sufficiently long. + + Packets with a supported version, or no Version field, are matched to + a connection using the connection ID or -- for packets with zero- + length connection IDs -- the local address and port. These packets + are processed using the selected connection; otherwise, the server + continues as described below. + + If the packet is an Initial packet fully conforming with the + specification, the server proceeds with the handshake (Section 7). + This commits the server to the version that the client selected. + + If a server refuses to accept a new connection, it SHOULD send an + Initial packet containing a CONNECTION_CLOSE frame with error code + CONNECTION_REFUSED. + + If the packet is a 0-RTT packet, the server MAY buffer a limited + number of these packets in anticipation of a late-arriving Initial + packet. Clients are not able to send Handshake packets prior to + receiving a server response, so servers SHOULD ignore any such + packets. + + Servers MUST drop incoming packets under all other circumstances. + +5.2.3. Considerations for Simple Load Balancers + + A server deployment could load-balance among servers using only + source and destination IP addresses and ports. Changes to the + client's IP address or port could result in packets being forwarded + to the wrong server. Such a server deployment could use one of the + following methods for connection continuity when a client's address + changes. + + * Servers could use an out-of-band mechanism to forward packets to + the correct server based on connection ID. + + * If servers can use a dedicated server IP address or port, other + than the one that the client initially connects to, they could use + the preferred_address transport parameter to request that clients + move connections to that dedicated address. Note that clients + could choose not to use the preferred address. + + A server in a deployment that does not implement a solution to + maintain connection continuity when the client address changes SHOULD + indicate that migration is not supported by using the + disable_active_migration transport parameter. The + disable_active_migration transport parameter does not prohibit + connection migration after a client has acted on a preferred_address + transport parameter. + + Server deployments that use this simple form of load balancing MUST + avoid the creation of a stateless reset oracle; see Section 21.11. + +5.3. Operations on Connections + + This document does not define an API for QUIC; it instead defines a + set of functions for QUIC connections that application protocols can + rely upon. An application protocol can assume that an implementation + of QUIC provides an interface that includes the operations described + in this section. An implementation designed for use with a specific + application protocol might provide only those operations that are + used by that protocol. + + When implementing the client role, an application protocol can: + + * open a connection, which begins the exchange described in + Section 7; + + * enable Early Data when available; and + + * be informed when Early Data has been accepted or rejected by a + server. + + When implementing the server role, an application protocol can: + + * listen for incoming connections, which prepares for the exchange + described in Section 7; + + * if Early Data is supported, embed application-controlled data in + the TLS resumption ticket sent to the client; and + + * if Early Data is supported, retrieve application-controlled data + from the client's resumption ticket and accept or reject Early + Data based on that information. + + In either role, an application protocol can: + + * configure minimum values for the initial number of permitted + streams of each type, as communicated in the transport parameters + (Section 7.4); + + * control resource allocation for receive buffers by setting flow + control limits both for streams and for the connection; + + * identify whether the handshake has completed successfully or is + still ongoing; + + * keep a connection from silently closing, by either generating PING + frames (Section 19.2) or requesting that the transport send + additional frames before the idle timeout expires (Section 10.1); + and + + * immediately close (Section 10.2) the connection. + +6. Version Negotiation + + Version negotiation allows a server to indicate that it does not + support the version the client used. A server sends a Version + Negotiation packet in response to each packet that might initiate a + new connection; see Section 5.2 for details. + + The size of the first packet sent by a client will determine whether + a server sends a Version Negotiation packet. Clients that support + multiple QUIC versions SHOULD ensure that the first UDP datagram they + send is sized to the largest of the minimum datagram sizes from all + versions they support, using PADDING frames (Section 19.1) as + necessary. This ensures that the server responds if there is a + mutually supported version. A server might not send a Version + Negotiation packet if the datagram it receives is smaller than the + minimum size specified in a different version; see Section 14.1. + +6.1. Sending Version Negotiation Packets + + If the version selected by the client is not acceptable to the + server, the server responds with a Version Negotiation packet; see + Section 17.2.1. This includes a list of versions that the server + will accept. An endpoint MUST NOT send a Version Negotiation packet + in response to receiving a Version Negotiation packet. + + This system allows a server to process packets with unsupported + versions without retaining state. Though either the Initial packet + or the Version Negotiation packet that is sent in response could be + lost, the client will send new packets until it successfully receives + a response or it abandons the connection attempt. + + A server MAY limit the number of Version Negotiation packets it + sends. For instance, a server that is able to recognize packets as + 0-RTT might choose not to send Version Negotiation packets in + response to 0-RTT packets with the expectation that it will + eventually receive an Initial packet. + +6.2. Handling Version Negotiation Packets + + Version Negotiation packets are designed to allow for functionality + to be defined in the future that allows QUIC to negotiate the version + of QUIC to use for a connection. Future Standards Track + specifications might change how implementations that support multiple + versions of QUIC react to Version Negotiation packets received in + response to an attempt to establish a connection using this version. + + A client that supports only this version of QUIC MUST abandon the + current connection attempt if it receives a Version Negotiation + packet, with the following two exceptions. A client MUST discard any + Version Negotiation packet if it has received and successfully + processed any other packet, including an earlier Version Negotiation + packet. A client MUST discard a Version Negotiation packet that + lists the QUIC version selected by the client. + + How to perform version negotiation is left as future work defined by + future Standards Track specifications. In particular, that future + work will ensure robustness against version downgrade attacks; see + Section 21.12. + +6.3. Using Reserved Versions + + For a server to use a new version in the future, clients need to + correctly handle unsupported versions. Some version numbers + (0x?a?a?a?a, as defined in Section 15) are reserved for inclusion in + fields that contain version numbers. + + Endpoints MAY add reserved versions to any field where unknown or + unsupported versions are ignored to test that a peer correctly + ignores the value. For instance, an endpoint could include a + reserved version in a Version Negotiation packet; see Section 17.2.1. + Endpoints MAY send packets with a reserved version to test that a + peer correctly discards the packet. + +7. Cryptographic and Transport Handshake + + QUIC relies on a combined cryptographic and transport handshake to + minimize connection establishment latency. QUIC uses the CRYPTO + frame (Section 19.6) to transmit the cryptographic handshake. The + version of QUIC defined in this document is identified as 0x00000001 + and uses TLS as described in [QUIC-TLS]; a different QUIC version + could indicate that a different cryptographic handshake protocol is + in use. + + QUIC provides reliable, ordered delivery of the cryptographic + handshake data. QUIC packet protection is used to encrypt as much of + the handshake protocol as possible. The cryptographic handshake MUST + provide the following properties: + + * authenticated key exchange, where + + - a server is always authenticated, + + - a client is optionally authenticated, + + - every connection produces distinct and unrelated keys, and + + - keying material is usable for packet protection for both 0-RTT + and 1-RTT packets. + + * authenticated exchange of values for transport parameters of both + endpoints, and confidentiality protection for server transport + parameters (see Section 7.4). + + * authenticated negotiation of an application protocol (TLS uses + Application-Layer Protocol Negotiation (ALPN) [ALPN] for this + purpose). + + The CRYPTO frame can be sent in different packet number spaces + (Section 12.3). The offsets used by CRYPTO frames to ensure ordered + delivery of cryptographic handshake data start from zero in each + packet number space. + + Figure 4 shows a simplified handshake and the exchange of packets and + frames that are used to advance the handshake. Exchange of + application data during the handshake is enabled where possible, + shown with an asterisk ("*"). Once the handshake is complete, + endpoints are able to exchange application data freely. + + Client Server + + Initial (CRYPTO) + 0-RTT (*) ----------> + Initial (CRYPTO) + Handshake (CRYPTO) + <---------- 1-RTT (*) + Handshake (CRYPTO) + 1-RTT (*) ----------> + <---------- 1-RTT (HANDSHAKE_DONE) + + 1-RTT <=========> 1-RTT + + Figure 4: Simplified QUIC Handshake + + Endpoints can use packets sent during the handshake to test for + Explicit Congestion Notification (ECN) support; see Section 13.4. An + endpoint validates support for ECN by observing whether the ACK + frames acknowledging the first packets it sends carry ECN counts, as + described in Section 13.4.2. + + Endpoints MUST explicitly negotiate an application protocol. This + avoids situations where there is a disagreement about the protocol + that is in use. + +7.1. Example Handshake Flows + + Details of how TLS is integrated with QUIC are provided in + [QUIC-TLS], but some examples are provided here. An extension of + this exchange to support client address validation is shown in + Section 8.1.2. + + Once any address validation exchanges are complete, the cryptographic + handshake is used to agree on cryptographic keys. The cryptographic + handshake is carried in Initial (Section 17.2.2) and Handshake + (Section 17.2.4) packets. + + Figure 5 provides an overview of the 1-RTT handshake. Each line + shows a QUIC packet with the packet type and packet number shown + first, followed by the frames that are typically contained in those + packets. For instance, the first packet is of type Initial, with + packet number 0, and contains a CRYPTO frame carrying the + ClientHello. + + Multiple QUIC packets -- even of different packet types -- can be + coalesced into a single UDP datagram; see Section 12.2. As a result, + this handshake could consist of as few as four UDP datagrams, or any + number more (subject to limits inherent to the protocol, such as + congestion control and anti-amplification). For instance, the + server's first flight contains Initial packets, Handshake packets, + and "0.5-RTT data" in 1-RTT packets. + + Client Server + + Initial[0]: CRYPTO[CH] -> + + Initial[0]: CRYPTO[SH] ACK[0] + Handshake[0]: CRYPTO[EE, CERT, CV, FIN] + <- 1-RTT[0]: STREAM[1, "..."] + + Initial[1]: ACK[0] + Handshake[0]: CRYPTO[FIN], ACK[0] + 1-RTT[0]: STREAM[0, "..."], ACK[0] -> + + Handshake[1]: ACK[0] + <- 1-RTT[1]: HANDSHAKE_DONE, STREAM[3, "..."], ACK[0] + + Figure 5: Example 1-RTT Handshake + + Figure 6 shows an example of a connection with a 0-RTT handshake and + a single packet of 0-RTT data. Note that as described in + Section 12.3, the server acknowledges 0-RTT data in 1-RTT packets, + and the client sends 1-RTT packets in the same packet number space. + + Client Server + + Initial[0]: CRYPTO[CH] + 0-RTT[0]: STREAM[0, "..."] -> + + Initial[0]: CRYPTO[SH] ACK[0] + Handshake[0] CRYPTO[EE, FIN] + <- 1-RTT[0]: STREAM[1, "..."] ACK[0] + + Initial[1]: ACK[0] + Handshake[0]: CRYPTO[FIN], ACK[0] + 1-RTT[1]: STREAM[0, "..."] ACK[0] -> + + Handshake[1]: ACK[0] + <- 1-RTT[1]: HANDSHAKE_DONE, STREAM[3, "..."], ACK[1] + + Figure 6: Example 0-RTT Handshake + +7.2. Negotiating Connection IDs + + A connection ID is used to ensure consistent routing of packets, as + described in Section 5.1. The long header contains two connection + IDs: the Destination Connection ID is chosen by the recipient of the + packet and is used to provide consistent routing; the Source + Connection ID is used to set the Destination Connection ID used by + the peer. + + During the handshake, packets with the long header (Section 17.2) are + used to establish the connection IDs used by both endpoints. Each + endpoint uses the Source Connection ID field to specify the + connection ID that is used in the Destination Connection ID field of + packets being sent to them. After processing the first Initial + packet, each endpoint sets the Destination Connection ID field in + subsequent packets it sends to the value of the Source Connection ID + field that it received. + + When an Initial packet is sent by a client that has not previously + received an Initial or Retry packet from the server, the client + populates the Destination Connection ID field with an unpredictable + value. This Destination Connection ID MUST be at least 8 bytes in + length. Until a packet is received from the server, the client MUST + use the same Destination Connection ID value on all packets in this + connection. + + The Destination Connection ID field from the first Initial packet + sent by a client is used to determine packet protection keys for + Initial packets. These keys change after receiving a Retry packet; + see Section 5.2 of [QUIC-TLS]. + + The client populates the Source Connection ID field with a value of + its choosing and sets the Source Connection ID Length field to + indicate the length. + + 0-RTT packets in the first flight use the same Destination Connection + ID and Source Connection ID values as the client's first Initial + packet. + + Upon first receiving an Initial or Retry packet from the server, the + client uses the Source Connection ID supplied by the server as the + Destination Connection ID for subsequent packets, including any 0-RTT + packets. This means that a client might have to change the + connection ID it sets in the Destination Connection ID field twice + during connection establishment: once in response to a Retry packet + and once in response to an Initial packet from the server. Once a + client has received a valid Initial packet from the server, it MUST + discard any subsequent packet it receives on that connection with a + different Source Connection ID. + + A client MUST change the Destination Connection ID it uses for + sending packets in response to only the first received Initial or + Retry packet. A server MUST set the Destination Connection ID it + uses for sending packets based on the first received Initial packet. + Any further changes to the Destination Connection ID are only + permitted if the values are taken from NEW_CONNECTION_ID frames; if + subsequent Initial packets include a different Source Connection ID, + they MUST be discarded. This avoids unpredictable outcomes that + might otherwise result from stateless processing of multiple Initial + packets with different Source Connection IDs. + + The Destination Connection ID that an endpoint sends can change over + the lifetime of a connection, especially in response to connection + migration (Section 9); see Section 5.1.1 for details. + +7.3. Authenticating Connection IDs + + The choice each endpoint makes about connection IDs during the + handshake is authenticated by including all values in transport + parameters; see Section 7.4. This ensures that all connection IDs + used for the handshake are also authenticated by the cryptographic + handshake. + + Each endpoint includes the value of the Source Connection ID field + from the first Initial packet it sent in the + initial_source_connection_id transport parameter; see Section 18.2. + A server includes the Destination Connection ID field from the first + Initial packet it received from the client in the + original_destination_connection_id transport parameter; if the server + sent a Retry packet, this refers to the first Initial packet received + before sending the Retry packet. If it sends a Retry packet, a + server also includes the Source Connection ID field from the Retry + packet in the retry_source_connection_id transport parameter. + + The values provided by a peer for these transport parameters MUST + match the values that an endpoint used in the Destination and Source + Connection ID fields of Initial packets that it sent (and received, + for servers). Endpoints MUST validate that received transport + parameters match received connection ID values. Including connection + ID values in transport parameters and verifying them ensures that an + attacker cannot influence the choice of connection ID for a + successful connection by injecting packets carrying attacker-chosen + connection IDs during the handshake. + + An endpoint MUST treat the absence of the + initial_source_connection_id transport parameter from either endpoint + or the absence of the original_destination_connection_id transport + parameter from the server as a connection error of type + TRANSPORT_PARAMETER_ERROR. + + An endpoint MUST treat the following as a connection error of type + TRANSPORT_PARAMETER_ERROR or PROTOCOL_VIOLATION: + + * absence of the retry_source_connection_id transport parameter from + the server after receiving a Retry packet, + + * presence of the retry_source_connection_id transport parameter + when no Retry packet was received, or + + * a mismatch between values received from a peer in these transport + parameters and the value sent in the corresponding Destination or + Source Connection ID fields of Initial packets. + + If a zero-length connection ID is selected, the corresponding + transport parameter is included with a zero-length value. + + Figure 7 shows the connection IDs (with DCID=Destination Connection + ID, SCID=Source Connection ID) that are used in a complete handshake. + The exchange of Initial packets is shown, plus the later exchange of + 1-RTT packets that includes the connection ID established during the + handshake. + + Client Server + + Initial: DCID=S1, SCID=C1 -> + <- Initial: DCID=C1, SCID=S3 + ... + 1-RTT: DCID=S3 -> + <- 1-RTT: DCID=C1 + + Figure 7: Use of Connection IDs in a Handshake + + Figure 8 shows a similar handshake that includes a Retry packet. + + Client Server + + Initial: DCID=S1, SCID=C1 -> + <- Retry: DCID=C1, SCID=S2 + Initial: DCID=S2, SCID=C1 -> + <- Initial: DCID=C1, SCID=S3 + ... + 1-RTT: DCID=S3 -> + <- 1-RTT: DCID=C1 + + Figure 8: Use of Connection IDs in a Handshake with Retry + + In both cases (Figures 7 and 8), the client sets the value of the + initial_source_connection_id transport parameter to "C1". + + When the handshake does not include a Retry (Figure 7), the server + sets original_destination_connection_id to "S1" (note that this value + is chosen by the client) and initial_source_connection_id to "S3". + In this case, the server does not include a + retry_source_connection_id transport parameter. + + When the handshake includes a Retry (Figure 8), the server sets + original_destination_connection_id to "S1", + retry_source_connection_id to "S2", and initial_source_connection_id + to "S3". + +7.4. Transport Parameters + + During connection establishment, both endpoints make authenticated + declarations of their transport parameters. Endpoints are required + to comply with the restrictions that each parameter defines; the + description of each parameter includes rules for its handling. + + Transport parameters are declarations that are made unilaterally by + each endpoint. Each endpoint can choose values for transport + parameters independent of the values chosen by its peer. + + The encoding of the transport parameters is detailed in Section 18. + + QUIC includes the encoded transport parameters in the cryptographic + handshake. Once the handshake completes, the transport parameters + declared by the peer are available. Each endpoint validates the + values provided by its peer. + + Definitions for each of the defined transport parameters are included + in Section 18.2. + + An endpoint MUST treat receipt of a transport parameter with an + invalid value as a connection error of type + TRANSPORT_PARAMETER_ERROR. + + An endpoint MUST NOT send a parameter more than once in a given + transport parameters extension. An endpoint SHOULD treat receipt of + duplicate transport parameters as a connection error of type + TRANSPORT_PARAMETER_ERROR. + + Endpoints use transport parameters to authenticate the negotiation of + connection IDs during the handshake; see Section 7.3. + + ALPN (see [ALPN]) allows clients to offer multiple application + protocols during connection establishment. The transport parameters + that a client includes during the handshake apply to all application + protocols that the client offers. Application protocols can + recommend values for transport parameters, such as the initial flow + control limits. However, application protocols that set constraints + on values for transport parameters could make it impossible for a + client to offer multiple application protocols if these constraints + conflict. + +7.4.1. Values of Transport Parameters for 0-RTT + + Using 0-RTT depends on both client and server using protocol + parameters that were negotiated from a previous connection. To + enable 0-RTT, endpoints store the values of the server transport + parameters with any session tickets it receives on the connection. + Endpoints also store any information required by the application + protocol or cryptographic handshake; see Section 4.6 of [QUIC-TLS]. + The values of stored transport parameters are used when attempting + 0-RTT using the session tickets. + + Remembered transport parameters apply to the new connection until the + handshake completes and the client starts sending 1-RTT packets. + Once the handshake completes, the client uses the transport + parameters established in the handshake. Not all transport + parameters are remembered, as some do not apply to future connections + or they have no effect on the use of 0-RTT. + + The definition of a new transport parameter (Section 7.4.2) MUST + specify whether storing the transport parameter for 0-RTT is + mandatory, optional, or prohibited. A client need not store a + transport parameter it cannot process. + + A client MUST NOT use remembered values for the following parameters: + ack_delay_exponent, max_ack_delay, initial_source_connection_id, + original_destination_connection_id, preferred_address, + retry_source_connection_id, and stateless_reset_token. The client + MUST use the server's new values in the handshake instead; if the + server does not provide new values, the default values are used. + + A client that attempts to send 0-RTT data MUST remember all other + transport parameters used by the server that it is able to process. + The server can remember these transport parameters or can store an + integrity-protected copy of the values in the ticket and recover the + information when accepting 0-RTT data. A server uses the transport + parameters in determining whether to accept 0-RTT data. + + If 0-RTT data is accepted by the server, the server MUST NOT reduce + any limits or alter any values that might be violated by the client + with its 0-RTT data. In particular, a server that accepts 0-RTT data + MUST NOT set values for the following parameters (Section 18.2) that + are smaller than the remembered values of the parameters. + + * active_connection_id_limit + + * initial_max_data + + * initial_max_stream_data_bidi_local + + * initial_max_stream_data_bidi_remote + + * initial_max_stream_data_uni + + * initial_max_streams_bidi + + * initial_max_streams_uni + + Omitting or setting a zero value for certain transport parameters can + result in 0-RTT data being enabled but not usable. The applicable + subset of transport parameters that permit the sending of application + data SHOULD be set to non-zero values for 0-RTT. This includes + initial_max_data and either (1) initial_max_streams_bidi and + initial_max_stream_data_bidi_remote or (2) initial_max_streams_uni + and initial_max_stream_data_uni. + + A server might provide larger initial stream flow control limits for + streams than the remembered values that a client applies when sending + 0-RTT. Once the handshake completes, the client updates the flow + control limits on all sending streams using the updated values of + initial_max_stream_data_bidi_remote and initial_max_stream_data_uni. + + A server MAY store and recover the previously sent values of the + max_idle_timeout, max_udp_payload_size, and disable_active_migration + parameters and reject 0-RTT if it selects smaller values. Lowering + the values of these parameters while also accepting 0-RTT data could + degrade the performance of the connection. Specifically, lowering + the max_udp_payload_size could result in dropped packets, leading to + worse performance compared to rejecting 0-RTT data outright. + + A server MUST reject 0-RTT data if the restored values for transport + parameters cannot be supported. + + When sending frames in 0-RTT packets, a client MUST only use + remembered transport parameters; importantly, it MUST NOT use updated + values that it learns from the server's updated transport parameters + or from frames received in 1-RTT packets. Updated values of + transport parameters from the handshake apply only to 1-RTT packets. + For instance, flow control limits from remembered transport + parameters apply to all 0-RTT packets even if those values are + increased by the handshake or by frames sent in 1-RTT packets. A + server MAY treat the use of updated transport parameters in 0-RTT as + a connection error of type PROTOCOL_VIOLATION. + +7.4.2. New Transport Parameters + + New transport parameters can be used to negotiate new protocol + behavior. An endpoint MUST ignore transport parameters that it does + not support. The absence of a transport parameter therefore disables + any optional protocol feature that is negotiated using the parameter. + As described in Section 18.1, some identifiers are reserved in order + to exercise this requirement. + + A client that does not understand a transport parameter can discard + it and attempt 0-RTT on subsequent connections. However, if the + client adds support for a discarded transport parameter, it risks + violating the constraints that the transport parameter establishes if + it attempts 0-RTT. New transport parameters can avoid this problem + by setting a default of the most conservative value. Clients can + avoid this problem by remembering all parameters, even those not + currently supported. + + New transport parameters can be registered according to the rules in + Section 22.3. + +7.5. Cryptographic Message Buffering + + Implementations need to maintain a buffer of CRYPTO data received out + of order. Because there is no flow control of CRYPTO frames, an + endpoint could potentially force its peer to buffer an unbounded + amount of data. + + Implementations MUST support buffering at least 4096 bytes of data + received in out-of-order CRYPTO frames. Endpoints MAY choose to + allow more data to be buffered during the handshake. A larger limit + during the handshake could allow for larger keys or credentials to be + exchanged. An endpoint's buffer size does not need to remain + constant during the life of the connection. + + Being unable to buffer CRYPTO frames during the handshake can lead to + a connection failure. If an endpoint's buffer is exceeded during the + handshake, it can expand its buffer temporarily to complete the + handshake. If an endpoint does not expand its buffer, it MUST close + the connection with a CRYPTO_BUFFER_EXCEEDED error code. + + Once the handshake completes, if an endpoint is unable to buffer all + data in a CRYPTO frame, it MAY discard that CRYPTO frame and all + CRYPTO frames received in the future, or it MAY close the connection + with a CRYPTO_BUFFER_EXCEEDED error code. Packets containing + discarded CRYPTO frames MUST be acknowledged because the packet has + been received and processed by the transport even though the CRYPTO + frame was discarded. + +8. Address Validation + + Address validation ensures that an endpoint cannot be used for a + traffic amplification attack. In such an attack, a packet is sent to + a server with spoofed source address information that identifies a + victim. If a server generates more or larger packets in response to + that packet, the attacker can use the server to send more data toward + the victim than it would be able to send on its own. + + The primary defense against amplification attacks is verifying that a + peer is able to receive packets at the transport address that it + claims. Therefore, after receiving packets from an address that is + not yet validated, an endpoint MUST limit the amount of data it sends + to the unvalidated address to three times the amount of data received + from that address. This limit on the size of responses is known as + the anti-amplification limit. + + Address validation is performed both during connection establishment + (see Section 8.1) and during connection migration (see Section 8.2). + +8.1. Address Validation during Connection Establishment + + Connection establishment implicitly provides address validation for + both endpoints. In particular, receipt of a packet protected with + Handshake keys confirms that the peer successfully processed an + Initial packet. Once an endpoint has successfully processed a + Handshake packet from the peer, it can consider the peer address to + have been validated. + + Additionally, an endpoint MAY consider the peer address validated if + the peer uses a connection ID chosen by the endpoint and the + connection ID contains at least 64 bits of entropy. + + For the client, the value of the Destination Connection ID field in + its first Initial packet allows it to validate the server address as + a part of successfully processing any packet. Initial packets from + the server are protected with keys that are derived from this value + (see Section 5.2 of [QUIC-TLS]). Alternatively, the value is echoed + by the server in Version Negotiation packets (Section 6) or included + in the Integrity Tag in Retry packets (Section 5.8 of [QUIC-TLS]). + + Prior to validating the client address, servers MUST NOT send more + than three times as many bytes as the number of bytes they have + received. This limits the magnitude of any amplification attack that + can be mounted using spoofed source addresses. For the purposes of + avoiding amplification prior to address validation, servers MUST + count all of the payload bytes received in datagrams that are + uniquely attributed to a single connection. This includes datagrams + that contain packets that are successfully processed and datagrams + that contain packets that are all discarded. + + Clients MUST ensure that UDP datagrams containing Initial packets + have UDP payloads of at least 1200 bytes, adding PADDING frames as + necessary. A client that sends padded datagrams allows the server to + send more data prior to completing address validation. + + Loss of an Initial or Handshake packet from the server can cause a + deadlock if the client does not send additional Initial or Handshake + packets. A deadlock could occur when the server reaches its anti- + amplification limit and the client has received acknowledgments for + all the data it has sent. In this case, when the client has no + reason to send additional packets, the server will be unable to send + more data because it has not validated the client's address. To + prevent this deadlock, clients MUST send a packet on a Probe Timeout + (PTO); see Section 6.2 of [QUIC-RECOVERY]. Specifically, the client + MUST send an Initial packet in a UDP datagram that contains at least + 1200 bytes if it does not have Handshake keys, and otherwise send a + Handshake packet. + + A server might wish to validate the client address before starting + the cryptographic handshake. QUIC uses a token in the Initial packet + to provide address validation prior to completing the handshake. + This token is delivered to the client during connection establishment + with a Retry packet (see Section 8.1.2) or in a previous connection + using the NEW_TOKEN frame (see Section 8.1.3). + + In addition to sending limits imposed prior to address validation, + servers are also constrained in what they can send by the limits set + by the congestion controller. Clients are only constrained by the + congestion controller. + +8.1.1. Token Construction + + A token sent in a NEW_TOKEN frame or a Retry packet MUST be + constructed in a way that allows the server to identify how it was + provided to a client. These tokens are carried in the same field but + require different handling from servers. + +8.1.2. Address Validation Using Retry Packets + + Upon receiving the client's Initial packet, the server can request + address validation by sending a Retry packet (Section 17.2.5) + containing a token. This token MUST be repeated by the client in all + Initial packets it sends for that connection after it receives the + Retry packet. + + In response to processing an Initial packet containing a token that + was provided in a Retry packet, a server cannot send another Retry + packet; it can only refuse the connection or permit it to proceed. + + As long as it is not possible for an attacker to generate a valid + token for its own address (see Section 8.1.4) and the client is able + to return that token, it proves to the server that it received the + token. + + A server can also use a Retry packet to defer the state and + processing costs of connection establishment. Requiring the server + to provide a different connection ID, along with the + original_destination_connection_id transport parameter defined in + Section 18.2, forces the server to demonstrate that it, or an entity + it cooperates with, received the original Initial packet from the + client. Providing a different connection ID also grants a server + some control over how subsequent packets are routed. This can be + used to direct connections to a different server instance. + + If a server receives a client Initial that contains an invalid Retry + token but is otherwise valid, it knows the client will not accept + another Retry token. The server can discard such a packet and allow + the client to time out to detect handshake failure, but that could + impose a significant latency penalty on the client. Instead, the + server SHOULD immediately close (Section 10.2) the connection with an + INVALID_TOKEN error. Note that a server has not established any + state for the connection at this point and so does not enter the + closing period. + + A flow showing the use of a Retry packet is shown in Figure 9. + + Client Server + + Initial[0]: CRYPTO[CH] -> + + <- Retry+Token + + Initial+Token[1]: CRYPTO[CH] -> + + Initial[0]: CRYPTO[SH] ACK[1] + Handshake[0]: CRYPTO[EE, CERT, CV, FIN] + <- 1-RTT[0]: STREAM[1, "..."] + + Figure 9: Example Handshake with Retry + +8.1.3. Address Validation for Future Connections + + A server MAY provide clients with an address validation token during + one connection that can be used on a subsequent connection. Address + validation is especially important with 0-RTT because a server + potentially sends a significant amount of data to a client in + response to 0-RTT data. + + The server uses the NEW_TOKEN frame (Section 19.7) to provide the + client with an address validation token that can be used to validate + future connections. In a future connection, the client includes this + token in Initial packets to provide address validation. The client + MUST include the token in all Initial packets it sends, unless a + Retry replaces the token with a newer one. The client MUST NOT use + the token provided in a Retry for future connections. Servers MAY + discard any Initial packet that does not carry the expected token. + + Unlike the token that is created for a Retry packet, which is used + immediately, the token sent in the NEW_TOKEN frame can be used after + some period of time has passed. Thus, a token SHOULD have an + expiration time, which could be either an explicit expiration time or + an issued timestamp that can be used to dynamically calculate the + expiration time. A server can store the expiration time or include + it in an encrypted form in the token. + + A token issued with NEW_TOKEN MUST NOT include information that would + allow values to be linked by an observer to the connection on which + it was issued. For example, it cannot include the previous + connection ID or addressing information, unless the values are + encrypted. A server MUST ensure that every NEW_TOKEN frame it sends + is unique across all clients, with the exception of those sent to + repair losses of previously sent NEW_TOKEN frames. Information that + allows the server to distinguish between tokens from Retry and + NEW_TOKEN MAY be accessible to entities other than the server. + + It is unlikely that the client port number is the same on two + different connections; validating the port is therefore unlikely to + be successful. + + A token received in a NEW_TOKEN frame is applicable to any server + that the connection is considered authoritative for (e.g., server + names included in the certificate). When connecting to a server for + which the client retains an applicable and unused token, it SHOULD + include that token in the Token field of its Initial packet. + Including a token might allow the server to validate the client + address without an additional round trip. A client MUST NOT include + a token that is not applicable to the server that it is connecting + to, unless the client has the knowledge that the server that issued + the token and the server the client is connecting to are jointly + managing the tokens. A client MAY use a token from any previous + connection to that server. + + A token allows a server to correlate activity between the connection + where the token was issued and any connection where it is used. + Clients that want to break continuity of identity with a server can + discard tokens provided using the NEW_TOKEN frame. In comparison, a + token obtained in a Retry packet MUST be used immediately during the + connection attempt and cannot be used in subsequent connection + attempts. + + A client SHOULD NOT reuse a token from a NEW_TOKEN frame for + different connection attempts. Reusing a token allows connections to + be linked by entities on the network path; see Section 9.5. + + Clients might receive multiple tokens on a single connection. Aside + from preventing linkability, any token can be used in any connection + attempt. Servers can send additional tokens to either enable address + validation for multiple connection attempts or replace older tokens + that might become invalid. For a client, this ambiguity means that + sending the most recent unused token is most likely to be effective. + Though saving and using older tokens have no negative consequences, + clients can regard older tokens as being less likely to be useful to + the server for address validation. + + When a server receives an Initial packet with an address validation + token, it MUST attempt to validate the token, unless it has already + completed address validation. If the token is invalid, then the + server SHOULD proceed as if the client did not have a validated + address, including potentially sending a Retry packet. Tokens + provided with NEW_TOKEN frames and Retry packets can be distinguished + by servers (see Section 8.1.1), and the latter can be validated more + strictly. If the validation succeeds, the server SHOULD then allow + the handshake to proceed. + + | Note: The rationale for treating the client as unvalidated + | rather than discarding the packet is that the client might have + | received the token in a previous connection using the NEW_TOKEN + | frame, and if the server has lost state, it might be unable to + | validate the token at all, leading to connection failure if the + | packet is discarded. + + In a stateless design, a server can use encrypted and authenticated + tokens to pass information to clients that the server can later + recover and use to validate a client address. Tokens are not + integrated into the cryptographic handshake, and so they are not + authenticated. For instance, a client might be able to reuse a + token. To avoid attacks that exploit this property, a server can + limit its use of tokens to only the information needed to validate + client addresses. + + Clients MAY use tokens obtained on one connection for any connection + attempt using the same version. When selecting a token to use, + clients do not need to consider other properties of the connection + that is being attempted, including the choice of possible application + protocols, session tickets, or other connection properties. + +8.1.4. Address Validation Token Integrity + + An address validation token MUST be difficult to guess. Including a + random value with at least 128 bits of entropy in the token would be + sufficient, but this depends on the server remembering the value it + sends to clients. + + A token-based scheme allows the server to offload any state + associated with validation to the client. For this design to work, + the token MUST be covered by integrity protection against + modification or falsification by clients. Without integrity + protection, malicious clients could generate or guess values for + tokens that would be accepted by the server. Only the server + requires access to the integrity protection key for tokens. + + There is no need for a single well-defined format for the token + because the server that generates the token also consumes it. Tokens + sent in Retry packets SHOULD include information that allows the + server to verify that the source IP address and port in client + packets remain constant. + + Tokens sent in NEW_TOKEN frames MUST include information that allows + the server to verify that the client IP address has not changed from + when the token was issued. Servers can use tokens from NEW_TOKEN + frames in deciding not to send a Retry packet, even if the client + address has changed. If the client IP address has changed, the + server MUST adhere to the anti-amplification limit; see Section 8. + Note that in the presence of NAT, this requirement might be + insufficient to protect other hosts that share the NAT from + amplification attacks. + + Attackers could replay tokens to use servers as amplifiers in DDoS + attacks. To protect against such attacks, servers MUST ensure that + replay of tokens is prevented or limited. Servers SHOULD ensure that + tokens sent in Retry packets are only accepted for a short time, as + they are returned immediately by clients. Tokens that are provided + in NEW_TOKEN frames (Section 19.7) need to be valid for longer but + SHOULD NOT be accepted multiple times. Servers are encouraged to + allow tokens to be used only once, if possible; tokens MAY include + additional information about clients to further narrow applicability + or reuse. + +8.2. Path Validation + + Path validation is used by both peers during connection migration + (see Section 9) to verify reachability after a change of address. In + path validation, endpoints test reachability between a specific local + address and a specific peer address, where an address is the 2-tuple + of IP address and port. + + Path validation tests that packets sent on a path to a peer are + received by that peer. Path validation is used to ensure that + packets received from a migrating peer do not carry a spoofed source + address. + + Path validation does not validate that a peer can send in the return + direction. Acknowledgments cannot be used for return path validation + because they contain insufficient entropy and might be spoofed. + Endpoints independently determine reachability on each direction of a + path, and therefore return reachability can only be established by + the peer. + + Path validation can be used at any time by either endpoint. For + instance, an endpoint might check that a peer is still in possession + of its address after a period of quiescence. + + Path validation is not designed as a NAT traversal mechanism. Though + the mechanism described here might be effective for the creation of + NAT bindings that support NAT traversal, the expectation is that one + endpoint is able to receive packets without first having sent a + packet on that path. Effective NAT traversal needs additional + synchronization mechanisms that are not provided here. + + An endpoint MAY include other frames with the PATH_CHALLENGE and + PATH_RESPONSE frames used for path validation. In particular, an + endpoint can include PADDING frames with a PATH_CHALLENGE frame for + Path Maximum Transmission Unit Discovery (PMTUD); see Section 14.2.1. + An endpoint can also include its own PATH_CHALLENGE frame when + sending a PATH_RESPONSE frame. + + An endpoint uses a new connection ID for probes sent from a new local + address; see Section 9.5. When probing a new path, an endpoint can + ensure that its peer has an unused connection ID available for + responses. Sending NEW_CONNECTION_ID and PATH_CHALLENGE frames in + the same packet, if the peer's active_connection_id_limit permits, + ensures that an unused connection ID will be available to the peer + when sending a response. + + An endpoint can choose to simultaneously probe multiple paths. The + number of simultaneous paths used for probes is limited by the number + of extra connection IDs its peer has previously supplied, since each + new local address used for a probe requires a previously unused + connection ID. + +8.2.1. Initiating Path Validation + + To initiate path validation, an endpoint sends a PATH_CHALLENGE frame + containing an unpredictable payload on the path to be validated. + + An endpoint MAY send multiple PATH_CHALLENGE frames to guard against + packet loss. However, an endpoint SHOULD NOT send multiple + PATH_CHALLENGE frames in a single packet. + + An endpoint SHOULD NOT probe a new path with packets containing a + PATH_CHALLENGE frame more frequently than it would send an Initial + packet. This ensures that connection migration is no more load on a + new path than establishing a new connection. + + The endpoint MUST use unpredictable data in every PATH_CHALLENGE + frame so that it can associate the peer's response with the + corresponding PATH_CHALLENGE. + + An endpoint MUST expand datagrams that contain a PATH_CHALLENGE frame + to at least the smallest allowed maximum datagram size of 1200 bytes, + unless the anti-amplification limit for the path does not permit + sending a datagram of this size. Sending UDP datagrams of this size + ensures that the network path from the endpoint to the peer can be + used for QUIC; see Section 14. + + When an endpoint is unable to expand the datagram size to 1200 bytes + due to the anti-amplification limit, the path MTU will not be + validated. To ensure that the path MTU is large enough, the endpoint + MUST perform a second path validation by sending a PATH_CHALLENGE + frame in a datagram of at least 1200 bytes. This additional + validation can be performed after a PATH_RESPONSE is successfully + received or when enough bytes have been received on the path that + sending the larger datagram will not result in exceeding the anti- + amplification limit. + + Unlike other cases where datagrams are expanded, endpoints MUST NOT + discard datagrams that appear to be too small when they contain + PATH_CHALLENGE or PATH_RESPONSE. + +8.2.2. Path Validation Responses + + On receiving a PATH_CHALLENGE frame, an endpoint MUST respond by + echoing the data contained in the PATH_CHALLENGE frame in a + PATH_RESPONSE frame. An endpoint MUST NOT delay transmission of a + packet containing a PATH_RESPONSE frame unless constrained by + congestion control. + + A PATH_RESPONSE frame MUST be sent on the network path where the + PATH_CHALLENGE frame was received. This ensures that path validation + by a peer only succeeds if the path is functional in both directions. + This requirement MUST NOT be enforced by the endpoint that initiates + path validation, as that would enable an attack on migration; see + Section 9.3.3. + + An endpoint MUST expand datagrams that contain a PATH_RESPONSE frame + to at least the smallest allowed maximum datagram size of 1200 bytes. + This verifies that the path is able to carry datagrams of this size + in both directions. However, an endpoint MUST NOT expand the + datagram containing the PATH_RESPONSE if the resulting data exceeds + the anti-amplification limit. This is expected to only occur if the + received PATH_CHALLENGE was not sent in an expanded datagram. + + An endpoint MUST NOT send more than one PATH_RESPONSE frame in + response to one PATH_CHALLENGE frame; see Section 13.3. The peer is + expected to send more PATH_CHALLENGE frames as necessary to evoke + additional PATH_RESPONSE frames. + +8.2.3. Successful Path Validation + + Path validation succeeds when a PATH_RESPONSE frame is received that + contains the data that was sent in a previous PATH_CHALLENGE frame. + A PATH_RESPONSE frame received on any network path validates the path + on which the PATH_CHALLENGE was sent. + + If an endpoint sends a PATH_CHALLENGE frame in a datagram that is not + expanded to at least 1200 bytes and if the response to it validates + the peer address, the path is validated but not the path MTU. As a + result, the endpoint can now send more than three times the amount of + data that has been received. However, the endpoint MUST initiate + another path validation with an expanded datagram to verify that the + path supports the required MTU. + + Receipt of an acknowledgment for a packet containing a PATH_CHALLENGE + frame is not adequate validation, since the acknowledgment can be + spoofed by a malicious peer. + +8.2.4. Failed Path Validation + + Path validation only fails when the endpoint attempting to validate + the path abandons its attempt to validate the path. + + Endpoints SHOULD abandon path validation based on a timer. When + setting this timer, implementations are cautioned that the new path + could have a longer round-trip time than the original. A value of + three times the larger of the current PTO or the PTO for the new path + (using kInitialRtt, as defined in [QUIC-RECOVERY]) is RECOMMENDED. + + This timeout allows for multiple PTOs to expire prior to failing path + validation, so that loss of a single PATH_CHALLENGE or PATH_RESPONSE + frame does not cause path validation failure. + + Note that the endpoint might receive packets containing other frames + on the new path, but a PATH_RESPONSE frame with appropriate data is + required for path validation to succeed. + + When an endpoint abandons path validation, it determines that the + path is unusable. This does not necessarily imply a failure of the + connection -- endpoints can continue sending packets over other paths + as appropriate. If no paths are available, an endpoint can wait for + a new path to become available or close the connection. An endpoint + that has no valid network path to its peer MAY signal this using the + NO_VIABLE_PATH connection error, noting that this is only possible if + the network path exists but does not support the required MTU + (Section 14). + + A path validation might be abandoned for other reasons besides + failure. Primarily, this happens if a connection migration to a new + path is initiated while a path validation on the old path is in + progress. + +9. Connection Migration + + The use of a connection ID allows connections to survive changes to + endpoint addresses (IP address and port), such as those caused by an + endpoint migrating to a new network. This section describes the + process by which an endpoint migrates to a new address. + + The design of QUIC relies on endpoints retaining a stable address for + the duration of the handshake. An endpoint MUST NOT initiate + connection migration before the handshake is confirmed, as defined in + Section 4.1.2 of [QUIC-TLS]. + + If the peer sent the disable_active_migration transport parameter, an + endpoint also MUST NOT send packets (including probing packets; see + Section 9.1) from a different local address to the address the peer + used during the handshake, unless the endpoint has acted on a + preferred_address transport parameter from the peer. If the peer + violates this requirement, the endpoint MUST either drop the incoming + packets on that path without generating a Stateless Reset or proceed + with path validation and allow the peer to migrate. Generating a + Stateless Reset or closing the connection would allow third parties + in the network to cause connections to close by spoofing or otherwise + manipulating observed traffic. + + Not all changes of peer address are intentional, or active, + migrations. The peer could experience NAT rebinding: a change of + address due to a middlebox, usually a NAT, allocating a new outgoing + port or even a new outgoing IP address for a flow. An endpoint MUST + perform path validation (Section 8.2) if it detects any change to a + peer's address, unless it has previously validated that address. + + When an endpoint has no validated path on which to send packets, it + MAY discard connection state. An endpoint capable of connection + migration MAY wait for a new path to become available before + discarding connection state. + + This document limits migration of connections to new client + addresses, except as described in Section 9.6. Clients are + responsible for initiating all migrations. Servers do not send non- + probing packets (see Section 9.1) toward a client address until they + see a non-probing packet from that address. If a client receives + packets from an unknown server address, the client MUST discard these + packets. + +9.1. Probing a New Path + + An endpoint MAY probe for peer reachability from a new local address + using path validation (Section 8.2) prior to migrating the connection + to the new local address. Failure of path validation simply means + that the new path is not usable for this connection. Failure to + validate a path does not cause the connection to end unless there are + no valid alternative paths available. + + PATH_CHALLENGE, PATH_RESPONSE, NEW_CONNECTION_ID, and PADDING frames + are "probing frames", and all other frames are "non-probing frames". + A packet containing only probing frames is a "probing packet", and a + packet containing any other frame is a "non-probing packet". + +9.2. Initiating Connection Migration + + An endpoint can migrate a connection to a new local address by + sending packets containing non-probing frames from that address. + + Each endpoint validates its peer's address during connection + establishment. Therefore, a migrating endpoint can send to its peer + knowing that the peer is willing to receive at the peer's current + address. Thus, an endpoint can migrate to a new local address + without first validating the peer's address. + + To establish reachability on the new path, an endpoint initiates path + validation (Section 8.2) on the new path. An endpoint MAY defer path + validation until after a peer sends the next non-probing frame to its + new address. + + When migrating, the new path might not support the endpoint's current + sending rate. Therefore, the endpoint resets its congestion + controller and RTT estimate, as described in Section 9.4. + + The new path might not have the same ECN capability. Therefore, the + endpoint validates ECN capability as described in Section 13.4. + +9.3. Responding to Connection Migration + + Receiving a packet from a new peer address containing a non-probing + frame indicates that the peer has migrated to that address. + + If the recipient permits the migration, it MUST send subsequent + packets to the new peer address and MUST initiate path validation + (Section 8.2) to verify the peer's ownership of the address if + validation is not already underway. If the recipient has no unused + connection IDs from the peer, it will not be able to send anything on + the new path until the peer provides one; see Section 9.5. + + An endpoint only changes the address to which it sends packets in + response to the highest-numbered non-probing packet. This ensures + that an endpoint does not send packets to an old peer address in the + case that it receives reordered packets. + + An endpoint MAY send data to an unvalidated peer address, but it MUST + protect against potential attacks as described in Sections 9.3.1 and + 9.3.2. An endpoint MAY skip validation of a peer address if that + address has been seen recently. In particular, if an endpoint + returns to a previously validated path after detecting some form of + spurious migration, skipping address validation and restoring loss + detection and congestion state can reduce the performance impact of + the attack. + + After changing the address to which it sends non-probing packets, an + endpoint can abandon any path validation for other addresses. + + Receiving a packet from a new peer address could be the result of a + NAT rebinding at the peer. + + After verifying a new client address, the server SHOULD send new + address validation tokens (Section 8) to the client. + +9.3.1. Peer Address Spoofing + + It is possible that a peer is spoofing its source address to cause an + endpoint to send excessive amounts of data to an unwilling host. If + the endpoint sends significantly more data than the spoofing peer, + connection migration might be used to amplify the volume of data that + an attacker can generate toward a victim. + + As described in Section 9.3, an endpoint is required to validate a + peer's new address to confirm the peer's possession of the new + address. Until a peer's address is deemed valid, an endpoint limits + the amount of data it sends to that address; see Section 8. In the + absence of this limit, an endpoint risks being used for a denial-of- + service attack against an unsuspecting victim. + + If an endpoint skips validation of a peer address as described above, + it does not need to limit its sending rate. + +9.3.2. On-Path Address Spoofing + + An on-path attacker could cause a spurious connection migration by + copying and forwarding a packet with a spoofed address such that it + arrives before the original packet. The packet with the spoofed + address will be seen to come from a migrating connection, and the + original packet will be seen as a duplicate and dropped. After a + spurious migration, validation of the source address will fail + because the entity at the source address does not have the necessary + cryptographic keys to read or respond to the PATH_CHALLENGE frame + that is sent to it even if it wanted to. + + To protect the connection from failing due to such a spurious + migration, an endpoint MUST revert to using the last validated peer + address when validation of a new peer address fails. Additionally, + receipt of packets with higher packet numbers from the legitimate + peer address will trigger another connection migration. This will + cause the validation of the address of the spurious migration to be + abandoned, thus containing migrations initiated by the attacker + injecting a single packet. + + If an endpoint has no state about the last validated peer address, it + MUST close the connection silently by discarding all connection + state. This results in new packets on the connection being handled + generically. For instance, an endpoint MAY send a Stateless Reset in + response to any further incoming packets. + +9.3.3. Off-Path Packet Forwarding + + An off-path attacker that can observe packets might forward copies of + genuine packets to endpoints. If the copied packet arrives before + the genuine packet, this will appear as a NAT rebinding. Any genuine + packet will be discarded as a duplicate. If the attacker is able to + continue forwarding packets, it might be able to cause migration to a + path via the attacker. This places the attacker on-path, giving it + the ability to observe or drop all subsequent packets. + + This style of attack relies on the attacker using a path that has + approximately the same characteristics as the direct path between + endpoints. The attack is more reliable if relatively few packets are + sent or if packet loss coincides with the attempted attack. + + A non-probing packet received on the original path that increases the + maximum received packet number will cause the endpoint to move back + to that path. Eliciting packets on this path increases the + likelihood that the attack is unsuccessful. Therefore, mitigation of + this attack relies on triggering the exchange of packets. + + In response to an apparent migration, endpoints MUST validate the + previously active path using a PATH_CHALLENGE frame. This induces + the sending of new packets on that path. If the path is no longer + viable, the validation attempt will time out and fail; if the path is + viable but no longer desired, the validation will succeed but only + results in probing packets being sent on the path. + + An endpoint that receives a PATH_CHALLENGE on an active path SHOULD + send a non-probing packet in response. If the non-probing packet + arrives before any copy made by an attacker, this results in the + connection being migrated back to the original path. Any subsequent + migration to another path restarts this entire process. + + This defense is imperfect, but this is not considered a serious + problem. If the path via the attack is reliably faster than the + original path despite multiple attempts to use that original path, it + is not possible to distinguish between an attack and an improvement + in routing. + + An endpoint could also use heuristics to improve detection of this + style of attack. For instance, NAT rebinding is improbable if + packets were recently received on the old path; similarly, rebinding + is rare on IPv6 paths. Endpoints can also look for duplicated + packets. Conversely, a change in connection ID is more likely to + indicate an intentional migration rather than an attack. + +9.4. Loss Detection and Congestion Control + + The capacity available on the new path might not be the same as the + old path. Packets sent on the old path MUST NOT contribute to + congestion control or RTT estimation for the new path. + + On confirming a peer's ownership of its new address, an endpoint MUST + immediately reset the congestion controller and round-trip time + estimator for the new path to initial values (see Appendices A.3 and + B.3 of [QUIC-RECOVERY]) unless the only change in the peer's address + is its port number. Because port-only changes are commonly the + result of NAT rebinding or other middlebox activity, the endpoint MAY + instead retain its congestion control state and round-trip estimate + in those cases instead of reverting to initial values. In cases + where congestion control state retained from an old path is used on a + new path with substantially different characteristics, a sender could + transmit too aggressively until the congestion controller and the RTT + estimator have adapted. Generally, implementations are advised to be + cautious when using previous values on a new path. + + There could be apparent reordering at the receiver when an endpoint + sends data and probes from/to multiple addresses during the migration + period, since the two resulting paths could have different round-trip + times. A receiver of packets on multiple paths will still send ACK + frames covering all received packets. + + While multiple paths might be used during connection migration, a + single congestion control context and a single loss recovery context + (as described in [QUIC-RECOVERY]) could be adequate. For instance, + an endpoint might delay switching to a new congestion control context + until it is confirmed that an old path is no longer needed (such as + the case described in Section 9.3.3). + + A sender can make exceptions for probe packets so that their loss + detection is independent and does not unduly cause the congestion + controller to reduce its sending rate. An endpoint might set a + separate timer when a PATH_CHALLENGE is sent, which is canceled if + the corresponding PATH_RESPONSE is received. If the timer fires + before the PATH_RESPONSE is received, the endpoint might send a new + PATH_CHALLENGE and restart the timer for a longer period of time. + This timer SHOULD be set as described in Section 6.2.1 of + [QUIC-RECOVERY] and MUST NOT be more aggressive. + +9.5. Privacy Implications of Connection Migration + + Using a stable connection ID on multiple network paths would allow a + passive observer to correlate activity between those paths. An + endpoint that moves between networks might not wish to have their + activity correlated by any entity other than their peer, so different + connection IDs are used when sending from different local addresses, + as discussed in Section 5.1. For this to be effective, endpoints + need to ensure that connection IDs they provide cannot be linked by + any other entity. + + At any time, endpoints MAY change the Destination Connection ID they + transmit with to a value that has not been used on another path. + + An endpoint MUST NOT reuse a connection ID when sending from more + than one local address -- for example, when initiating connection + migration as described in Section 9.2 or when probing a new network + path as described in Section 9.1. + + Similarly, an endpoint MUST NOT reuse a connection ID when sending to + more than one destination address. Due to network changes outside + the control of its peer, an endpoint might receive packets from a new + source address with the same Destination Connection ID field value, + in which case it MAY continue to use the current connection ID with + the new remote address while still sending from the same local + address. + + These requirements regarding connection ID reuse apply only to the + sending of packets, as unintentional changes in path without a change + in connection ID are possible. For example, after a period of + network inactivity, NAT rebinding might cause packets to be sent on a + new path when the client resumes sending. An endpoint responds to + such an event as described in Section 9.3. + + Using different connection IDs for packets sent in both directions on + each new network path eliminates the use of the connection ID for + linking packets from the same connection across different network + paths. Header protection ensures that packet numbers cannot be used + to correlate activity. This does not prevent other properties of + packets, such as timing and size, from being used to correlate + activity. + + An endpoint SHOULD NOT initiate migration with a peer that has + requested a zero-length connection ID, because traffic over the new + path might be trivially linkable to traffic over the old one. If the + server is able to associate packets with a zero-length connection ID + to the right connection, it means that the server is using other + information to demultiplex packets. For example, a server might + provide a unique address to every client -- for instance, using HTTP + alternative services [ALTSVC]. Information that might allow correct + routing of packets across multiple network paths will also allow + activity on those paths to be linked by entities other than the peer. + + A client might wish to reduce linkability by switching to a new + connection ID, source UDP port, or IP address (see [RFC8981]) when + sending traffic after a period of inactivity. Changing the address + from which it sends packets at the same time might cause the server + to detect a connection migration. This ensures that the mechanisms + that support migration are exercised even for clients that do not + experience NAT rebindings or genuine migrations. Changing address + can cause a peer to reset its congestion control state (see + Section 9.4), so addresses SHOULD only be changed infrequently. + + An endpoint that exhausts available connection IDs cannot probe new + paths or initiate migration, nor can it respond to probes or attempts + by its peer to migrate. To ensure that migration is possible and + packets sent on different paths cannot be correlated, endpoints + SHOULD provide new connection IDs before peers migrate; see + Section 5.1.1. If a peer might have exhausted available connection + IDs, a migrating endpoint could include a NEW_CONNECTION_ID frame in + all packets sent on a new network path. + +9.6. Server's Preferred Address + + QUIC allows servers to accept connections on one IP address and + attempt to transfer these connections to a more preferred address + shortly after the handshake. This is particularly useful when + clients initially connect to an address shared by multiple servers + but would prefer to use a unicast address to ensure connection + stability. This section describes the protocol for migrating a + connection to a preferred server address. + + Migrating a connection to a new server address mid-connection is not + supported by the version of QUIC specified in this document. If a + client receives packets from a new server address when the client has + not initiated a migration to that address, the client SHOULD discard + these packets. + +9.6.1. Communicating a Preferred Address + + A server conveys a preferred address by including the + preferred_address transport parameter in the TLS handshake. + + Servers MAY communicate a preferred address of each address family + (IPv4 and IPv6) to allow clients to pick the one most suited to their + network attachment. + + Once the handshake is confirmed, the client SHOULD select one of the + two addresses provided by the server and initiate path validation + (see Section 8.2). A client constructs packets using any previously + unused active connection ID, taken from either the preferred_address + transport parameter or a NEW_CONNECTION_ID frame. + + As soon as path validation succeeds, the client SHOULD begin sending + all future packets to the new server address using the new connection + ID and discontinue use of the old server address. If path validation + fails, the client MUST continue sending all future packets to the + server's original IP address. + +9.6.2. Migration to a Preferred Address + + A client that migrates to a preferred address MUST validate the + address it chooses before migrating; see Section 21.5.3. + + A server might receive a packet addressed to its preferred IP address + at any time after it accepts a connection. If this packet contains a + PATH_CHALLENGE frame, the server sends a packet containing a + PATH_RESPONSE frame as per Section 8.2. The server MUST send non- + probing packets from its original address until it receives a non- + probing packet from the client at its preferred address and until the + server has validated the new path. + + The server MUST probe on the path toward the client from its + preferred address. This helps to guard against spurious migration + initiated by an attacker. + + Once the server has completed its path validation and has received a + non-probing packet with a new largest packet number on its preferred + address, the server begins sending non-probing packets to the client + exclusively from its preferred IP address. The server SHOULD drop + newer packets for this connection that are received on the old IP + address. The server MAY continue to process delayed packets that are + received on the old IP address. + + The addresses that a server provides in the preferred_address + transport parameter are only valid for the connection in which they + are provided. A client MUST NOT use these for other connections, + including connections that are resumed from the current connection. + +9.6.3. Interaction of Client Migration and Preferred Address + + A client might need to perform a connection migration before it has + migrated to the server's preferred address. In this case, the client + SHOULD perform path validation to both the original and preferred + server address from the client's new address concurrently. + + If path validation of the server's preferred address succeeds, the + client MUST abandon validation of the original address and migrate to + using the server's preferred address. If path validation of the + server's preferred address fails but validation of the server's + original address succeeds, the client MAY migrate to its new address + and continue sending to the server's original address. + + If packets received at the server's preferred address have a + different source address than observed from the client during the + handshake, the server MUST protect against potential attacks as + described in Sections 9.3.1 and 9.3.2. In addition to intentional + simultaneous migration, this might also occur because the client's + access network used a different NAT binding for the server's + preferred address. + + Servers SHOULD initiate path validation to the client's new address + upon receiving a probe packet from a different address; see + Section 8. + + A client that migrates to a new address SHOULD use a preferred + address from the same address family for the server. + + The connection ID provided in the preferred_address transport + parameter is not specific to the addresses that are provided. This + connection ID is provided to ensure that the client has a connection + ID available for migration, but the client MAY use this connection ID + on any path. + +9.7. Use of IPv6 Flow Label and Migration + + Endpoints that send data using IPv6 SHOULD apply an IPv6 flow label + in compliance with [RFC6437], unless the local API does not allow + setting IPv6 flow labels. + + The flow label generation MUST be designed to minimize the chances of + linkability with a previously used flow label, as a stable flow label + would enable correlating activity on multiple paths; see Section 9.5. + + [RFC6437] suggests deriving values using a pseudorandom function to + generate flow labels. Including the Destination Connection ID field + in addition to source and destination addresses when generating flow + labels ensures that changes are synchronized with changes in other + observable identifiers. A cryptographic hash function that combines + these inputs with a local secret is one way this might be + implemented. + +10. Connection Termination + + An established QUIC connection can be terminated in one of three + ways: + + * idle timeout (Section 10.1) + + * immediate close (Section 10.2) + + * stateless reset (Section 10.3) + + An endpoint MAY discard connection state if it does not have a + validated path on which it can send packets; see Section 8.2. + +10.1. Idle Timeout + + If a max_idle_timeout is specified by either endpoint in its + transport parameters (Section 18.2), the connection is silently + closed and its state is discarded when it remains idle for longer + than the minimum of the max_idle_timeout value advertised by both + endpoints. + + Each endpoint advertises a max_idle_timeout, but the effective value + at an endpoint is computed as the minimum of the two advertised + values (or the sole advertised value, if only one endpoint advertises + a non-zero value). By announcing a max_idle_timeout, an endpoint + commits to initiating an immediate close (Section 10.2) if it + abandons the connection prior to the effective value. + + An endpoint restarts its idle timer when a packet from its peer is + received and processed successfully. An endpoint also restarts its + idle timer when sending an ack-eliciting packet if no other ack- + eliciting packets have been sent since last receiving and processing + a packet. Restarting this timer when sending a packet ensures that + connections are not closed after new activity is initiated. + + To avoid excessively small idle timeout periods, endpoints MUST + increase the idle timeout period to be at least three times the + current Probe Timeout (PTO). This allows for multiple PTOs to + expire, and therefore multiple probes to be sent and lost, prior to + idle timeout. + +10.1.1. Liveness Testing + + An endpoint that sends packets close to the effective timeout risks + having them be discarded at the peer, since the idle timeout period + might have expired at the peer before these packets arrive. + + An endpoint can send a PING or another ack-eliciting frame to test + the connection for liveness if the peer could time out soon, such as + within a PTO; see Section 6.2 of [QUIC-RECOVERY]. This is especially + useful if any available application data cannot be safely retried. + Note that the application determines what data is safe to retry. + +10.1.2. Deferring Idle Timeout + + An endpoint might need to send ack-eliciting packets to avoid an idle + timeout if it is expecting response data but does not have or is + unable to send application data. + + An implementation of QUIC might provide applications with an option + to defer an idle timeout. This facility could be used when the + application wishes to avoid losing state that has been associated + with an open connection but does not expect to exchange application + data for some time. With this option, an endpoint could send a PING + frame (Section 19.2) periodically, which will cause the peer to + restart its idle timeout period. Sending a packet containing a PING + frame restarts the idle timeout for this endpoint also if this is the + first ack-eliciting packet sent since receiving a packet. Sending a + PING frame causes the peer to respond with an acknowledgment, which + also restarts the idle timeout for the endpoint. + + Application protocols that use QUIC SHOULD provide guidance on when + deferring an idle timeout is appropriate. Unnecessary sending of + PING frames could have a detrimental effect on performance. + + A connection will time out if no packets are sent or received for a + period longer than the time negotiated using the max_idle_timeout + transport parameter; see Section 10. However, state in middleboxes + might time out earlier than that. Though REQ-5 in [RFC4787] + recommends a 2-minute timeout interval, experience shows that sending + packets every 30 seconds is necessary to prevent the majority of + middleboxes from losing state for UDP flows [GATEWAY]. + +10.2. Immediate Close + + An endpoint sends a CONNECTION_CLOSE frame (Section 19.19) to + terminate the connection immediately. A CONNECTION_CLOSE frame + causes all streams to immediately become closed; open streams can be + assumed to be implicitly reset. + + After sending a CONNECTION_CLOSE frame, an endpoint immediately + enters the closing state; see Section 10.2.1. After receiving a + CONNECTION_CLOSE frame, endpoints enter the draining state; see + Section 10.2.2. + + Violations of the protocol lead to an immediate close. + + An immediate close can be used after an application protocol has + arranged to close a connection. This might be after the application + protocol negotiates a graceful shutdown. The application protocol + can exchange messages that are needed for both application endpoints + to agree that the connection can be closed, after which the + application requests that QUIC close the connection. When QUIC + consequently closes the connection, a CONNECTION_CLOSE frame with an + application-supplied error code will be used to signal closure to the + peer. + + The closing and draining connection states exist to ensure that + connections close cleanly and that delayed or reordered packets are + properly discarded. These states SHOULD persist for at least three + times the current PTO interval as defined in [QUIC-RECOVERY]. + + Disposing of connection state prior to exiting the closing or + draining state could result in an endpoint generating a Stateless + Reset unnecessarily when it receives a late-arriving packet. + Endpoints that have some alternative means to ensure that late- + arriving packets do not induce a response, such as those that are + able to close the UDP socket, MAY end these states earlier to allow + for faster resource recovery. Servers that retain an open socket for + accepting new connections SHOULD NOT end the closing or draining + state early. + + Once its closing or draining state ends, an endpoint SHOULD discard + all connection state. The endpoint MAY send a Stateless Reset in + response to any further incoming packets belonging to this + connection. + +10.2.1. Closing Connection State + + An endpoint enters the closing state after initiating an immediate + close. + + In the closing state, an endpoint retains only enough information to + generate a packet containing a CONNECTION_CLOSE frame and to identify + packets as belonging to the connection. An endpoint in the closing + state sends a packet containing a CONNECTION_CLOSE frame in response + to any incoming packet that it attributes to the connection. + + An endpoint SHOULD limit the rate at which it generates packets in + the closing state. For instance, an endpoint could wait for a + progressively increasing number of received packets or amount of time + before responding to received packets. + + An endpoint's selected connection ID and the QUIC version are + sufficient information to identify packets for a closing connection; + the endpoint MAY discard all other connection state. An endpoint + that is closing is not required to process any received frame. An + endpoint MAY retain packet protection keys for incoming packets to + allow it to read and process a CONNECTION_CLOSE frame. + + An endpoint MAY drop packet protection keys when entering the closing + state and send a packet containing a CONNECTION_CLOSE frame in + response to any UDP datagram that is received. However, an endpoint + that discards packet protection keys cannot identify and discard + invalid packets. To avoid being used for an amplification attack, + such endpoints MUST limit the cumulative size of packets it sends to + three times the cumulative size of the packets that are received and + attributed to the connection. To minimize the state that an endpoint + maintains for a closing connection, endpoints MAY send the exact same + packet in response to any received packet. + + | Note: Allowing retransmission of a closing packet is an + | exception to the requirement that a new packet number be used + | for each packet; see Section 12.3. Sending new packet numbers + | is primarily of advantage to loss recovery and congestion + | control, which are not expected to be relevant for a closed + | connection. Retransmitting the final packet requires less + | state. + + While in the closing state, an endpoint could receive packets from a + new source address, possibly indicating a connection migration; see + Section 9. An endpoint in the closing state MUST either discard + packets received from an unvalidated address or limit the cumulative + size of packets it sends to an unvalidated address to three times the + size of packets it receives from that address. + + An endpoint is not expected to handle key updates when it is closing + (Section 6 of [QUIC-TLS]). A key update might prevent the endpoint + from moving from the closing state to the draining state, as the + endpoint will not be able to process subsequently received packets, + but it otherwise has no impact. + +10.2.2. Draining Connection State + + The draining state is entered once an endpoint receives a + CONNECTION_CLOSE frame, which indicates that its peer is closing or + draining. While otherwise identical to the closing state, an + endpoint in the draining state MUST NOT send any packets. Retaining + packet protection keys is unnecessary once a connection is in the + draining state. + + An endpoint that receives a CONNECTION_CLOSE frame MAY send a single + packet containing a CONNECTION_CLOSE frame before entering the + draining state, using a NO_ERROR code if appropriate. An endpoint + MUST NOT send further packets. Doing so could result in a constant + exchange of CONNECTION_CLOSE frames until one of the endpoints exits + the closing state. + + An endpoint MAY enter the draining state from the closing state if it + receives a CONNECTION_CLOSE frame, which indicates that the peer is + also closing or draining. In this case, the draining state ends when + the closing state would have ended. In other words, the endpoint + uses the same end time but ceases transmission of any packets on this + connection. + +10.2.3. Immediate Close during the Handshake + + When sending a CONNECTION_CLOSE frame, the goal is to ensure that the + peer will process the frame. Generally, this means sending the frame + in a packet with the highest level of packet protection to avoid the + packet being discarded. After the handshake is confirmed (see + Section 4.1.2 of [QUIC-TLS]), an endpoint MUST send any + CONNECTION_CLOSE frames in a 1-RTT packet. However, prior to + confirming the handshake, it is possible that more advanced packet + protection keys are not available to the peer, so another + CONNECTION_CLOSE frame MAY be sent in a packet that uses a lower + packet protection level. More specifically: + + * A client will always know whether the server has Handshake keys + (see Section 17.2.2.1), but it is possible that a server does not + know whether the client has Handshake keys. Under these + circumstances, a server SHOULD send a CONNECTION_CLOSE frame in + both Handshake and Initial packets to ensure that at least one of + them is processable by the client. + + * A client that sends a CONNECTION_CLOSE frame in a 0-RTT packet + cannot be assured that the server has accepted 0-RTT. Sending a + CONNECTION_CLOSE frame in an Initial packet makes it more likely + that the server can receive the close signal, even if the + application error code might not be received. + + * Prior to confirming the handshake, a peer might be unable to + process 1-RTT packets, so an endpoint SHOULD send a + CONNECTION_CLOSE frame in both Handshake and 1-RTT packets. A + server SHOULD also send a CONNECTION_CLOSE frame in an Initial + packet. + + Sending a CONNECTION_CLOSE of type 0x1d in an Initial or Handshake + packet could expose application state or be used to alter application + state. A CONNECTION_CLOSE of type 0x1d MUST be replaced by a + CONNECTION_CLOSE of type 0x1c when sending the frame in Initial or + Handshake packets. Otherwise, information about the application + state might be revealed. Endpoints MUST clear the value of the + Reason Phrase field and SHOULD use the APPLICATION_ERROR code when + converting to a CONNECTION_CLOSE of type 0x1c. + + CONNECTION_CLOSE frames sent in multiple packet types can be + coalesced into a single UDP datagram; see Section 12.2. + + An endpoint can send a CONNECTION_CLOSE frame in an Initial packet. + This might be in response to unauthenticated information received in + Initial or Handshake packets. Such an immediate close might expose + legitimate connections to a denial of service. QUIC does not include + defensive measures for on-path attacks during the handshake; see + Section 21.2. However, at the cost of reducing feedback about errors + for legitimate peers, some forms of denial of service can be made + more difficult for an attacker if endpoints discard illegal packets + rather than terminating a connection with CONNECTION_CLOSE. For this + reason, endpoints MAY discard packets rather than immediately close + if errors are detected in packets that lack authentication. + + An endpoint that has not established state, such as a server that + detects an error in an Initial packet, does not enter the closing + state. An endpoint that has no state for the connection does not + enter a closing or draining period on sending a CONNECTION_CLOSE + frame. + +10.3. Stateless Reset + + A stateless reset is provided as an option of last resort for an + endpoint that does not have access to the state of a connection. A + crash or outage might result in peers continuing to send data to an + endpoint that is unable to properly continue the connection. An + endpoint MAY send a Stateless Reset in response to receiving a packet + that it cannot associate with an active connection. + + A stateless reset is not appropriate for indicating errors in active + connections. An endpoint that wishes to communicate a fatal + connection error MUST use a CONNECTION_CLOSE frame if it is able. + + To support this process, an endpoint issues a stateless reset token, + which is a 16-byte value that is hard to guess. If the peer + subsequently receives a Stateless Reset, which is a UDP datagram that + ends in that stateless reset token, the peer will immediately end the + connection. + + A stateless reset token is specific to a connection ID. An endpoint + issues a stateless reset token by including the value in the + Stateless Reset Token field of a NEW_CONNECTION_ID frame. Servers + can also issue a stateless_reset_token transport parameter during the + handshake that applies to the connection ID that it selected during + the handshake. These exchanges are protected by encryption, so only + client and server know their value. Note that clients cannot use the + stateless_reset_token transport parameter because their transport + parameters do not have confidentiality protection. + + Tokens are invalidated when their associated connection ID is retired + via a RETIRE_CONNECTION_ID frame (Section 19.16). + + An endpoint that receives packets that it cannot process sends a + packet in the following layout (see Section 1.3): + + Stateless Reset { + Fixed Bits (2) = 1, + Unpredictable Bits (38..), + Stateless Reset Token (128), + } + + Figure 10: Stateless Reset + + This design ensures that a Stateless Reset is -- to the extent + possible -- indistinguishable from a regular packet with a short + header. + + A Stateless Reset uses an entire UDP datagram, starting with the + first two bits of the packet header. The remainder of the first byte + and an arbitrary number of bytes following it are set to values that + SHOULD be indistinguishable from random. The last 16 bytes of the + datagram contain a stateless reset token. + + To entities other than its intended recipient, a Stateless Reset will + appear to be a packet with a short header. For the Stateless Reset + to appear as a valid QUIC packet, the Unpredictable Bits field needs + to include at least 38 bits of data (or 5 bytes, less the two fixed + bits). + + The resulting minimum size of 21 bytes does not guarantee that a + Stateless Reset is difficult to distinguish from other packets if the + recipient requires the use of a connection ID. To achieve that end, + the endpoint SHOULD ensure that all packets it sends are at least 22 + bytes longer than the minimum connection ID length that it requests + the peer to include in its packets, adding PADDING frames as + necessary. This ensures that any Stateless Reset sent by the peer is + indistinguishable from a valid packet sent to the endpoint. An + endpoint that sends a Stateless Reset in response to a packet that is + 43 bytes or shorter SHOULD send a Stateless Reset that is one byte + shorter than the packet it responds to. + + These values assume that the stateless reset token is the same length + as the minimum expansion of the packet protection AEAD. Additional + unpredictable bytes are necessary if the endpoint could have + negotiated a packet protection scheme with a larger minimum + expansion. + + An endpoint MUST NOT send a Stateless Reset that is three times or + more larger than the packet it receives to avoid being used for + amplification. Section 10.3.3 describes additional limits on + Stateless Reset size. + + Endpoints MUST discard packets that are too small to be valid QUIC + packets. To give an example, with the set of AEAD functions defined + in [QUIC-TLS], short header packets that are smaller than 21 bytes + are never valid. + + Endpoints MUST send Stateless Resets formatted as a packet with a + short header. However, endpoints MUST treat any packet ending in a + valid stateless reset token as a Stateless Reset, as other QUIC + versions might allow the use of a long header. + + An endpoint MAY send a Stateless Reset in response to a packet with a + long header. Sending a Stateless Reset is not effective prior to the + stateless reset token being available to a peer. In this QUIC + version, packets with a long header are only used during connection + establishment. Because the stateless reset token is not available + until connection establishment is complete or near completion, + ignoring an unknown packet with a long header might be as effective + as sending a Stateless Reset. + + An endpoint cannot determine the Source Connection ID from a packet + with a short header; therefore, it cannot set the Destination + Connection ID in the Stateless Reset. The Destination Connection ID + will therefore differ from the value used in previous packets. A + random Destination Connection ID makes the connection ID appear to be + the result of moving to a new connection ID that was provided using a + NEW_CONNECTION_ID frame; see Section 19.15. + + Using a randomized connection ID results in two problems: + + * The packet might not reach the peer. If the Destination + Connection ID is critical for routing toward the peer, then this + packet could be incorrectly routed. This might also trigger + another Stateless Reset in response; see Section 10.3.3. A + Stateless Reset that is not correctly routed is an ineffective + error detection and recovery mechanism. In this case, endpoints + will need to rely on other methods -- such as timers -- to detect + that the connection has failed. + + * The randomly generated connection ID can be used by entities other + than the peer to identify this as a potential Stateless Reset. An + endpoint that occasionally uses different connection IDs might + introduce some uncertainty about this. + + This stateless reset design is specific to QUIC version 1. An + endpoint that supports multiple versions of QUIC needs to generate a + Stateless Reset that will be accepted by peers that support any + version that the endpoint might support (or might have supported + prior to losing state). Designers of new versions of QUIC need to be + aware of this and either (1) reuse this design or (2) use a portion + of the packet other than the last 16 bytes for carrying data. + +10.3.1. Detecting a Stateless Reset + + An endpoint detects a potential Stateless Reset using the trailing 16 + bytes of the UDP datagram. An endpoint remembers all stateless reset + tokens associated with the connection IDs and remote addresses for + datagrams it has recently sent. This includes Stateless Reset Token + field values from NEW_CONNECTION_ID frames and the server's transport + parameters but excludes stateless reset tokens associated with + connection IDs that are either unused or retired. The endpoint + identifies a received datagram as a Stateless Reset by comparing the + last 16 bytes of the datagram with all stateless reset tokens + associated with the remote address on which the datagram was + received. + + This comparison can be performed for every inbound datagram. + Endpoints MAY skip this check if any packet from a datagram is + successfully processed. However, the comparison MUST be performed + when the first packet in an incoming datagram either cannot be + associated with a connection or cannot be decrypted. + + An endpoint MUST NOT check for any stateless reset tokens associated + with connection IDs it has not used or for connection IDs that have + been retired. + + When comparing a datagram to stateless reset token values, endpoints + MUST perform the comparison without leaking information about the + value of the token. For example, performing this comparison in + constant time protects the value of individual stateless reset tokens + from information leakage through timing side channels. Another + approach would be to store and compare the transformed values of + stateless reset tokens instead of the raw token values, where the + transformation is defined as a cryptographically secure pseudorandom + function using a secret key (e.g., block cipher, Hashed Message + Authentication Code (HMAC) [RFC2104]). An endpoint is not expected + to protect information about whether a packet was successfully + decrypted or the number of valid stateless reset tokens. + + If the last 16 bytes of the datagram are identical in value to a + stateless reset token, the endpoint MUST enter the draining period + and not send any further packets on this connection. + +10.3.2. Calculating a Stateless Reset Token + + The stateless reset token MUST be difficult to guess. In order to + create a stateless reset token, an endpoint could randomly generate + [RANDOM] a secret for every connection that it creates. However, + this presents a coordination problem when there are multiple + instances in a cluster or a storage problem for an endpoint that + might lose state. Stateless reset specifically exists to handle the + case where state is lost, so this approach is suboptimal. + + A single static key can be used across all connections to the same + endpoint by generating the proof using a pseudorandom function that + takes a static key and the connection ID chosen by the endpoint (see + Section 5.1) as input. An endpoint could use HMAC [RFC2104] (for + example, HMAC(static_key, connection_id)) or the HMAC-based Key + Derivation Function (HKDF) [RFC5869] (for example, using the static + key as input keying material, with the connection ID as salt). The + output of this function is truncated to 16 bytes to produce the + stateless reset token for that connection. + + An endpoint that loses state can use the same method to generate a + valid stateless reset token. The connection ID comes from the packet + that the endpoint receives. + + This design relies on the peer always sending a connection ID in its + packets so that the endpoint can use the connection ID from a packet + to reset the connection. An endpoint that uses this design MUST + either use the same connection ID length for all connections or + encode the length of the connection ID such that it can be recovered + without state. In addition, it cannot provide a zero-length + connection ID. + + Revealing the stateless reset token allows any entity to terminate + the connection, so a value can only be used once. This method for + choosing the stateless reset token means that the combination of + connection ID and static key MUST NOT be used for another connection. + A denial-of-service attack is possible if the same connection ID is + used by instances that share a static key or if an attacker can cause + a packet to be routed to an instance that has no state but the same + static key; see Section 21.11. A connection ID from a connection + that is reset by revealing the stateless reset token MUST NOT be + reused for new connections at nodes that share a static key. + + The same stateless reset token MUST NOT be used for multiple + connection IDs. Endpoints are not required to compare new values + against all previous values, but a duplicate value MAY be treated as + a connection error of type PROTOCOL_VIOLATION. + + Note that Stateless Resets do not have any cryptographic protection. + +10.3.3. Looping + + The design of a Stateless Reset is such that without knowing the + stateless reset token it is indistinguishable from a valid packet. + For instance, if a server sends a Stateless Reset to another server, + it might receive another Stateless Reset in response, which could + lead to an infinite exchange. + + An endpoint MUST ensure that every Stateless Reset that it sends is + smaller than the packet that triggered it, unless it maintains state + sufficient to prevent looping. In the event of a loop, this results + in packets eventually being too small to trigger a response. + + An endpoint can remember the number of Stateless Resets that it has + sent and stop generating new Stateless Resets once a limit is + reached. Using separate limits for different remote addresses will + ensure that Stateless Resets can be used to close connections when + other peers or connections have exhausted limits. + + A Stateless Reset that is smaller than 41 bytes might be identifiable + as a Stateless Reset by an observer, depending upon the length of the + peer's connection IDs. Conversely, not sending a Stateless Reset in + response to a small packet might result in Stateless Resets not being + useful in detecting cases of broken connections where only very small + packets are sent; such failures might only be detected by other + means, such as timers. + +11. Error Handling + + An endpoint that detects an error SHOULD signal the existence of that + error to its peer. Both transport-level and application-level errors + can affect an entire connection; see Section 11.1. Only application- + level errors can be isolated to a single stream; see Section 11.2. + + The most appropriate error code (Section 20) SHOULD be included in + the frame that signals the error. Where this specification + identifies error conditions, it also identifies the error code that + is used; though these are worded as requirements, different + implementation strategies might lead to different errors being + reported. In particular, an endpoint MAY use any applicable error + code when it detects an error condition; a generic error code (such + as PROTOCOL_VIOLATION or INTERNAL_ERROR) can always be used in place + of specific error codes. + + A stateless reset (Section 10.3) is not suitable for any error that + can be signaled with a CONNECTION_CLOSE or RESET_STREAM frame. A + stateless reset MUST NOT be used by an endpoint that has the state + necessary to send a frame on the connection. + +11.1. Connection Errors + + Errors that result in the connection being unusable, such as an + obvious violation of protocol semantics or corruption of state that + affects an entire connection, MUST be signaled using a + CONNECTION_CLOSE frame (Section 19.19). + + Application-specific protocol errors are signaled using the + CONNECTION_CLOSE frame with a frame type of 0x1d. Errors that are + specific to the transport, including all those described in this + document, are carried in the CONNECTION_CLOSE frame with a frame type + of 0x1c. + + A CONNECTION_CLOSE frame could be sent in a packet that is lost. An + endpoint SHOULD be prepared to retransmit a packet containing a + CONNECTION_CLOSE frame if it receives more packets on a terminated + connection. Limiting the number of retransmissions and the time over + which this final packet is sent limits the effort expended on + terminated connections. + + An endpoint that chooses not to retransmit packets containing a + CONNECTION_CLOSE frame risks a peer missing the first such packet. + The only mechanism available to an endpoint that continues to receive + data for a terminated connection is to attempt the stateless reset + process (Section 10.3). + + As the AEAD for Initial packets does not provide strong + authentication, an endpoint MAY discard an invalid Initial packet. + Discarding an Initial packet is permitted even where this + specification otherwise mandates a connection error. An endpoint can + only discard a packet if it does not process the frames in the packet + or reverts the effects of any processing. Discarding invalid Initial + packets might be used to reduce exposure to denial of service; see + Section 21.2. + +11.2. Stream Errors + + If an application-level error affects a single stream but otherwise + leaves the connection in a recoverable state, the endpoint can send a + RESET_STREAM frame (Section 19.4) with an appropriate error code to + terminate just the affected stream. + + Resetting a stream without the involvement of the application + protocol could cause the application protocol to enter an + unrecoverable state. RESET_STREAM MUST only be instigated by the + application protocol that uses QUIC. + + The semantics of the application error code carried in RESET_STREAM + are defined by the application protocol. Only the application + protocol is able to cause a stream to be terminated. A local + instance of the application protocol uses a direct API call, and a + remote instance uses the STOP_SENDING frame, which triggers an + automatic RESET_STREAM. + + Application protocols SHOULD define rules for handling streams that + are prematurely canceled by either endpoint. + +12. Packets and Frames + + QUIC endpoints communicate by exchanging packets. Packets have + confidentiality and integrity protection; see Section 12.1. Packets + are carried in UDP datagrams; see Section 12.2. + + This version of QUIC uses the long packet header during connection + establishment; see Section 17.2. Packets with the long header are + Initial (Section 17.2.2), 0-RTT (Section 17.2.3), Handshake + (Section 17.2.4), and Retry (Section 17.2.5). Version negotiation + uses a version-independent packet with a long header; see + Section 17.2.1. + + Packets with the short header are designed for minimal overhead and + are used after a connection is established and 1-RTT keys are + available; see Section 17.3. + +12.1. Protected Packets + + QUIC packets have different levels of cryptographic protection based + on the type of packet. Details of packet protection are found in + [QUIC-TLS]; this section includes an overview of the protections that + are provided. + + Version Negotiation packets have no cryptographic protection; see + [QUIC-INVARIANTS]. + + Retry packets use an AEAD function [AEAD] to protect against + accidental modification. + + Initial packets use an AEAD function, the keys for which are derived + using a value that is visible on the wire. Initial packets therefore + do not have effective confidentiality protection. Initial protection + exists to ensure that the sender of the packet is on the network + path. Any entity that receives an Initial packet from a client can + recover the keys that will allow them to both read the contents of + the packet and generate Initial packets that will be successfully + authenticated at either endpoint. The AEAD also protects Initial + packets against accidental modification. + + All other packets are protected with keys derived from the + cryptographic handshake. The cryptographic handshake ensures that + only the communicating endpoints receive the corresponding keys for + Handshake, 0-RTT, and 1-RTT packets. Packets protected with 0-RTT + and 1-RTT keys have strong confidentiality and integrity protection. + + The Packet Number field that appears in some packet types has + alternative confidentiality protection that is applied as part of + header protection; see Section 5.4 of [QUIC-TLS] for details. The + underlying packet number increases with each packet sent in a given + packet number space; see Section 12.3 for details. + +12.2. Coalescing Packets + + Initial (Section 17.2.2), 0-RTT (Section 17.2.3), and Handshake + (Section 17.2.4) packets contain a Length field that determines the + end of the packet. The length includes both the Packet Number and + Payload fields, both of which are confidentiality protected and + initially of unknown length. The length of the Payload field is + learned once header protection is removed. + + Using the Length field, a sender can coalesce multiple QUIC packets + into one UDP datagram. This can reduce the number of UDP datagrams + needed to complete the cryptographic handshake and start sending + data. This can also be used to construct Path Maximum Transmission + Unit (PMTU) probes; see Section 14.4.1. Receivers MUST be able to + process coalesced packets. + + Coalescing packets in order of increasing encryption levels (Initial, + 0-RTT, Handshake, 1-RTT; see Section 4.1.4 of [QUIC-TLS]) makes it + more likely that the receiver will be able to process all the packets + in a single pass. A packet with a short header does not include a + length, so it can only be the last packet included in a UDP datagram. + An endpoint SHOULD include multiple frames in a single packet if they + are to be sent at the same encryption level, instead of coalescing + multiple packets at the same encryption level. + + Receivers MAY route based on the information in the first packet + contained in a UDP datagram. Senders MUST NOT coalesce QUIC packets + with different connection IDs into a single UDP datagram. Receivers + SHOULD ignore any subsequent packets with a different Destination + Connection ID than the first packet in the datagram. + + Every QUIC packet that is coalesced into a single UDP datagram is + separate and complete. The receiver of coalesced QUIC packets MUST + individually process each QUIC packet and separately acknowledge + them, as if they were received as the payload of different UDP + datagrams. For example, if decryption fails (because the keys are + not available or for any other reason), the receiver MAY either + discard or buffer the packet for later processing and MUST attempt to + process the remaining packets. + + Retry packets (Section 17.2.5), Version Negotiation packets + (Section 17.2.1), and packets with a short header (Section 17.3) do + not contain a Length field and so cannot be followed by other packets + in the same UDP datagram. Note also that there is no situation where + a Retry or Version Negotiation packet is coalesced with another + packet. + +12.3. Packet Numbers + + The packet number is an integer in the range 0 to 2^62-1. This + number is used in determining the cryptographic nonce for packet + protection. Each endpoint maintains a separate packet number for + sending and receiving. + + Packet numbers are limited to this range because they need to be + representable in whole in the Largest Acknowledged field of an ACK + frame (Section 19.3). When present in a long or short header, + however, packet numbers are reduced and encoded in 1 to 4 bytes; see + Section 17.1. + + Version Negotiation (Section 17.2.1) and Retry (Section 17.2.5) + packets do not include a packet number. + + Packet numbers are divided into three spaces in QUIC: + + Initial space: All Initial packets (Section 17.2.2) are in this + space. + + Handshake space: All Handshake packets (Section 17.2.4) are in this + space. + + Application data space: All 0-RTT (Section 17.2.3) and 1-RTT + (Section 17.3.1) packets are in this space. + + As described in [QUIC-TLS], each packet type uses different + protection keys. + + Conceptually, a packet number space is the context in which a packet + can be processed and acknowledged. Initial packets can only be sent + with Initial packet protection keys and acknowledged in packets that + are also Initial packets. Similarly, Handshake packets are sent at + the Handshake encryption level and can only be acknowledged in + Handshake packets. + + This enforces cryptographic separation between the data sent in the + different packet number spaces. Packet numbers in each space start + at packet number 0. Subsequent packets sent in the same packet + number space MUST increase the packet number by at least one. + + 0-RTT and 1-RTT data exist in the same packet number space to make + loss recovery algorithms easier to implement between the two packet + types. + + A QUIC endpoint MUST NOT reuse a packet number within the same packet + number space in one connection. If the packet number for sending + reaches 2^62-1, the sender MUST close the connection without sending + a CONNECTION_CLOSE frame or any further packets; an endpoint MAY send + a Stateless Reset (Section 10.3) in response to further packets that + it receives. + + A receiver MUST discard a newly unprotected packet unless it is + certain that it has not processed another packet with the same packet + number from the same packet number space. Duplicate suppression MUST + happen after removing packet protection for the reasons described in + Section 9.5 of [QUIC-TLS]. + + Endpoints that track all individual packets for the purposes of + detecting duplicates are at risk of accumulating excessive state. + The data required for detecting duplicates can be limited by + maintaining a minimum packet number below which all packets are + immediately dropped. Any minimum needs to account for large + variations in round-trip time, which includes the possibility that a + peer might probe network paths with much larger round-trip times; see + Section 9. + + Packet number encoding at a sender and decoding at a receiver are + described in Section 17.1. + +12.4. Frames and Frame Types + + The payload of QUIC packets, after removing packet protection, + consists of a sequence of complete frames, as shown in Figure 11. + Version Negotiation, Stateless Reset, and Retry packets do not + contain frames. + + Packet Payload { + Frame (8..) ..., + } + + Figure 11: QUIC Payload + + The payload of a packet that contains frames MUST contain at least + one frame, and MAY contain multiple frames and multiple frame types. + An endpoint MUST treat receipt of a packet containing no frames as a + connection error of type PROTOCOL_VIOLATION. Frames always fit + within a single QUIC packet and cannot span multiple packets. + + Each frame begins with a Frame Type, indicating its type, followed by + additional type-dependent fields: + + Frame { + Frame Type (i), + Type-Dependent Fields (..), + } + + Figure 12: Generic Frame Layout + + Table 3 lists and summarizes information about each frame type that + is defined in this specification. A description of this summary is + included after the table. + + +============+======================+===============+======+======+ + | Type Value | Frame Type Name | Definition | Pkts | Spec | + +============+======================+===============+======+======+ + | 0x00 | PADDING | Section 19.1 | IH01 | NP | + +------------+----------------------+---------------+------+------+ + | 0x01 | PING | Section 19.2 | IH01 | | + +------------+----------------------+---------------+------+------+ + | 0x02-0x03 | ACK | Section 19.3 | IH_1 | NC | + +------------+----------------------+---------------+------+------+ + | 0x04 | RESET_STREAM | Section 19.4 | __01 | | + +------------+----------------------+---------------+------+------+ + | 0x05 | STOP_SENDING | Section 19.5 | __01 | | + +------------+----------------------+---------------+------+------+ + | 0x06 | CRYPTO | Section 19.6 | IH_1 | | + +------------+----------------------+---------------+------+------+ + | 0x07 | NEW_TOKEN | Section 19.7 | ___1 | | + +------------+----------------------+---------------+------+------+ + | 0x08-0x0f | STREAM | Section 19.8 | __01 | F | + +------------+----------------------+---------------+------+------+ + | 0x10 | MAX_DATA | Section 19.9 | __01 | | + +------------+----------------------+---------------+------+------+ + | 0x11 | MAX_STREAM_DATA | Section 19.10 | __01 | | + +------------+----------------------+---------------+------+------+ + | 0x12-0x13 | MAX_STREAMS | Section 19.11 | __01 | | + +------------+----------------------+---------------+------+------+ + | 0x14 | DATA_BLOCKED | Section 19.12 | __01 | | + +------------+----------------------+---------------+------+------+ + | 0x15 | STREAM_DATA_BLOCKED | Section 19.13 | __01 | | + +------------+----------------------+---------------+------+------+ + | 0x16-0x17 | STREAMS_BLOCKED | Section 19.14 | __01 | | + +------------+----------------------+---------------+------+------+ + | 0x18 | NEW_CONNECTION_ID | Section 19.15 | __01 | P | + +------------+----------------------+---------------+------+------+ + | 0x19 | RETIRE_CONNECTION_ID | Section 19.16 | __01 | | + +------------+----------------------+---------------+------+------+ + | 0x1a | PATH_CHALLENGE | Section 19.17 | __01 | P | + +------------+----------------------+---------------+------+------+ + | 0x1b | PATH_RESPONSE | Section 19.18 | ___1 | P | + +------------+----------------------+---------------+------+------+ + | 0x1c-0x1d | CONNECTION_CLOSE | Section 19.19 | ih01 | N | + +------------+----------------------+---------------+------+------+ + | 0x1e | HANDSHAKE_DONE | Section 19.20 | ___1 | | + +------------+----------------------+---------------+------+------+ + + Table 3: Frame Types + + The format and semantics of each frame type are explained in more + detail in Section 19. The remainder of this section provides a + summary of important and general information. + + The Frame Type in ACK, STREAM, MAX_STREAMS, STREAMS_BLOCKED, and + CONNECTION_CLOSE frames is used to carry other frame-specific flags. + For all other frames, the Frame Type field simply identifies the + frame. + + The "Pkts" column in Table 3 lists the types of packets that each + frame type could appear in, indicated by the following characters: + + I: Initial (Section 17.2.2) + + H: Handshake (Section 17.2.4) + + 0: 0-RTT (Section 17.2.3) + + 1: 1-RTT (Section 17.3.1) + + ih: Only a CONNECTION_CLOSE frame of type 0x1c can appear in Initial + or Handshake packets. + + For more details about these restrictions, see Section 12.5. Note + that all frames can appear in 1-RTT packets. An endpoint MUST treat + receipt of a frame in a packet type that is not permitted as a + connection error of type PROTOCOL_VIOLATION. + + The "Spec" column in Table 3 summarizes any special rules governing + the processing or generation of the frame type, as indicated by the + following characters: + + N: Packets containing only frames with this marking are not ack- + eliciting; see Section 13.2. + + C: Packets containing only frames with this marking do not count + toward bytes in flight for congestion control purposes; see + [QUIC-RECOVERY]. + + P: Packets containing only frames with this marking can be used to + probe new network paths during connection migration; see + Section 9.1. + + F: The contents of frames with this marking are flow controlled; + see Section 4. + + The "Pkts" and "Spec" columns in Table 3 do not form part of the IANA + registry; see Section 22.4. + + An endpoint MUST treat the receipt of a frame of unknown type as a + connection error of type FRAME_ENCODING_ERROR. + + All frames are idempotent in this version of QUIC. That is, a valid + frame does not cause undesirable side effects or errors when received + more than once. + + The Frame Type field uses a variable-length integer encoding (see + Section 16), with one exception. To ensure simple and efficient + implementations of frame parsing, a frame type MUST use the shortest + possible encoding. For frame types defined in this document, this + means a single-byte encoding, even though it is possible to encode + these values as a two-, four-, or eight-byte variable-length integer. + For instance, though 0x4001 is a legitimate two-byte encoding for a + variable-length integer with a value of 1, PING frames are always + encoded as a single byte with the value 0x01. This rule applies to + all current and future QUIC frame types. An endpoint MAY treat the + receipt of a frame type that uses a longer encoding than necessary as + a connection error of type PROTOCOL_VIOLATION. + +12.5. Frames and Number Spaces + + Some frames are prohibited in different packet number spaces. The + rules here generalize those of TLS, in that frames associated with + establishing the connection can usually appear in packets in any + packet number space, whereas those associated with transferring data + can only appear in the application data packet number space: + + * PADDING, PING, and CRYPTO frames MAY appear in any packet number + space. + + * CONNECTION_CLOSE frames signaling errors at the QUIC layer (type + 0x1c) MAY appear in any packet number space. CONNECTION_CLOSE + frames signaling application errors (type 0x1d) MUST only appear + in the application data packet number space. + + * ACK frames MAY appear in any packet number space but can only + acknowledge packets that appeared in that packet number space. + However, as noted below, 0-RTT packets cannot contain ACK frames. + + * All other frame types MUST only be sent in the application data + packet number space. + + Note that it is not possible to send the following frames in 0-RTT + packets for various reasons: ACK, CRYPTO, HANDSHAKE_DONE, NEW_TOKEN, + PATH_RESPONSE, and RETIRE_CONNECTION_ID. A server MAY treat receipt + of these frames in 0-RTT packets as a connection error of type + PROTOCOL_VIOLATION. + +13. Packetization and Reliability + + A sender sends one or more frames in a QUIC packet; see Section 12.4. + + A sender can minimize per-packet bandwidth and computational costs by + including as many frames as possible in each QUIC packet. A sender + MAY wait for a short period of time to collect multiple frames before + sending a packet that is not maximally packed, to avoid sending out + large numbers of small packets. An implementation MAY use knowledge + about application sending behavior or heuristics to determine whether + and for how long to wait. This waiting period is an implementation + decision, and an implementation should be careful to delay + conservatively, since any delay is likely to increase application- + visible latency. + + Stream multiplexing is achieved by interleaving STREAM frames from + multiple streams into one or more QUIC packets. A single QUIC packet + can include multiple STREAM frames from one or more streams. + + One of the benefits of QUIC is avoidance of head-of-line blocking + across multiple streams. When a packet loss occurs, only streams + with data in that packet are blocked waiting for a retransmission to + be received, while other streams can continue making progress. Note + that when data from multiple streams is included in a single QUIC + packet, loss of that packet blocks all those streams from making + progress. Implementations are advised to include as few streams as + necessary in outgoing packets without losing transmission efficiency + to underfilled packets. + +13.1. Packet Processing + + A packet MUST NOT be acknowledged until packet protection has been + successfully removed and all frames contained in the packet have been + processed. For STREAM frames, this means the data has been enqueued + in preparation to be received by the application protocol, but it + does not require that data be delivered and consumed. + + Once the packet has been fully processed, a receiver acknowledges + receipt by sending one or more ACK frames containing the packet + number of the received packet. + + An endpoint SHOULD treat receipt of an acknowledgment for a packet it + did not send as a connection error of type PROTOCOL_VIOLATION, if it + is able to detect the condition. For further discussion of how this + might be achieved, see Section 21.4. + +13.2. Generating Acknowledgments + + Endpoints acknowledge all packets they receive and process. However, + only ack-eliciting packets cause an ACK frame to be sent within the + maximum ack delay. Packets that are not ack-eliciting are only + acknowledged when an ACK frame is sent for other reasons. + + When sending a packet for any reason, an endpoint SHOULD attempt to + include an ACK frame if one has not been sent recently. Doing so + helps with timely loss detection at the peer. + + In general, frequent feedback from a receiver improves loss and + congestion response, but this has to be balanced against excessive + load generated by a receiver that sends an ACK frame in response to + every ack-eliciting packet. The guidance offered below seeks to + strike this balance. + +13.2.1. Sending ACK Frames + + Every packet SHOULD be acknowledged at least once, and ack-eliciting + packets MUST be acknowledged at least once within the maximum delay + an endpoint communicated using the max_ack_delay transport parameter; + see Section 18.2. max_ack_delay declares an explicit contract: an + endpoint promises to never intentionally delay acknowledgments of an + ack-eliciting packet by more than the indicated value. If it does, + any excess accrues to the RTT estimate and could result in spurious + or delayed retransmissions from the peer. A sender uses the + receiver's max_ack_delay value in determining timeouts for timer- + based retransmission, as detailed in Section 6.2 of [QUIC-RECOVERY]. + + An endpoint MUST acknowledge all ack-eliciting Initial and Handshake + packets immediately and all ack-eliciting 0-RTT and 1-RTT packets + within its advertised max_ack_delay, with the following exception. + Prior to handshake confirmation, an endpoint might not have packet + protection keys for decrypting Handshake, 0-RTT, or 1-RTT packets + when they are received. It might therefore buffer them and + acknowledge them when the requisite keys become available. + + Since packets containing only ACK frames are not congestion + controlled, an endpoint MUST NOT send more than one such packet in + response to receiving an ack-eliciting packet. + + An endpoint MUST NOT send a non-ack-eliciting packet in response to a + non-ack-eliciting packet, even if there are packet gaps that precede + the received packet. This avoids an infinite feedback loop of + acknowledgments, which could prevent the connection from ever + becoming idle. Non-ack-eliciting packets are eventually acknowledged + when the endpoint sends an ACK frame in response to other events. + + An endpoint that is only sending ACK frames will not receive + acknowledgments from its peer unless those acknowledgments are + included in packets with ack-eliciting frames. An endpoint SHOULD + send an ACK frame with other frames when there are new ack-eliciting + packets to acknowledge. When only non-ack-eliciting packets need to + be acknowledged, an endpoint MAY choose not to send an ACK frame with + outgoing frames until an ack-eliciting packet has been received. + + An endpoint that is only sending non-ack-eliciting packets might + choose to occasionally add an ack-eliciting frame to those packets to + ensure that it receives an acknowledgment; see Section 13.2.4. In + that case, an endpoint MUST NOT send an ack-eliciting frame in all + packets that would otherwise be non-ack-eliciting, to avoid an + infinite feedback loop of acknowledgments. + + In order to assist loss detection at the sender, an endpoint SHOULD + generate and send an ACK frame without delay when it receives an ack- + eliciting packet either: + + * when the received packet has a packet number less than another + ack-eliciting packet that has been received, or + + * when the packet has a packet number larger than the highest- + numbered ack-eliciting packet that has been received and there are + missing packets between that packet and this packet. + + Similarly, packets marked with the ECN Congestion Experienced (CE) + codepoint in the IP header SHOULD be acknowledged immediately, to + reduce the peer's response time to congestion events. + + The algorithms in [QUIC-RECOVERY] are expected to be resilient to + receivers that do not follow the guidance offered above. However, an + implementation should only deviate from these requirements after + careful consideration of the performance implications of a change, + for connections made by the endpoint and for other users of the + network. + +13.2.2. Acknowledgment Frequency + + A receiver determines how frequently to send acknowledgments in + response to ack-eliciting packets. This determination involves a + trade-off. + + Endpoints rely on timely acknowledgment to detect loss; see Section 6 + of [QUIC-RECOVERY]. Window-based congestion controllers, such as the + one described in Section 7 of [QUIC-RECOVERY], rely on + acknowledgments to manage their congestion window. In both cases, + delaying acknowledgments can adversely affect performance. + + On the other hand, reducing the frequency of packets that carry only + acknowledgments reduces packet transmission and processing cost at + both endpoints. It can improve connection throughput on severely + asymmetric links and reduce the volume of acknowledgment traffic + using return path capacity; see Section 3 of [RFC3449]. + + A receiver SHOULD send an ACK frame after receiving at least two ack- + eliciting packets. This recommendation is general in nature and + consistent with recommendations for TCP endpoint behavior [RFC5681]. + Knowledge of network conditions, knowledge of the peer's congestion + controller, or further research and experimentation might suggest + alternative acknowledgment strategies with better performance + characteristics. + + A receiver MAY process multiple available packets before determining + whether to send an ACK frame in response. + +13.2.3. Managing ACK Ranges + + When an ACK frame is sent, one or more ranges of acknowledged packets + are included. Including acknowledgments for older packets reduces + the chance of spurious retransmissions caused by losing previously + sent ACK frames, at the cost of larger ACK frames. + + ACK frames SHOULD always acknowledge the most recently received + packets, and the more out of order the packets are, the more + important it is to send an updated ACK frame quickly, to prevent the + peer from declaring a packet as lost and spuriously retransmitting + the frames it contains. An ACK frame is expected to fit within a + single QUIC packet. If it does not, then older ranges (those with + the smallest packet numbers) are omitted. + + A receiver limits the number of ACK Ranges (Section 19.3.1) it + remembers and sends in ACK frames, both to limit the size of ACK + frames and to avoid resource exhaustion. After receiving + acknowledgments for an ACK frame, the receiver SHOULD stop tracking + those acknowledged ACK Ranges. Senders can expect acknowledgments + for most packets, but QUIC does not guarantee receipt of an + acknowledgment for every packet that the receiver processes. + + It is possible that retaining many ACK Ranges could cause an ACK + frame to become too large. A receiver can discard unacknowledged ACK + Ranges to limit ACK frame size, at the cost of increased + retransmissions from the sender. This is necessary if an ACK frame + would be too large to fit in a packet. Receivers MAY also limit ACK + frame size further to preserve space for other frames or to limit the + capacity that acknowledgments consume. + + A receiver MUST retain an ACK Range unless it can ensure that it will + not subsequently accept packets with numbers in that range. + Maintaining a minimum packet number that increases as ranges are + discarded is one way to achieve this with minimal state. + + Receivers can discard all ACK Ranges, but they MUST retain the + largest packet number that has been successfully processed, as that + is used to recover packet numbers from subsequent packets; see + Section 17.1. + + A receiver SHOULD include an ACK Range containing the largest + received packet number in every ACK frame. The Largest Acknowledged + field is used in ECN validation at a sender, and including a lower + value than what was included in a previous ACK frame could cause ECN + to be unnecessarily disabled; see Section 13.4.2. + + Section 13.2.4 describes an exemplary approach for determining what + packets to acknowledge in each ACK frame. Though the goal of this + algorithm is to generate an acknowledgment for every packet that is + processed, it is still possible for acknowledgments to be lost. + +13.2.4. Limiting Ranges by Tracking ACK Frames + + When a packet containing an ACK frame is sent, the Largest + Acknowledged field in that frame can be saved. When a packet + containing an ACK frame is acknowledged, the receiver can stop + acknowledging packets less than or equal to the Largest Acknowledged + field in the sent ACK frame. + + A receiver that sends only non-ack-eliciting packets, such as ACK + frames, might not receive an acknowledgment for a long period of + time. This could cause the receiver to maintain state for a large + number of ACK frames for a long period of time, and ACK frames it + sends could be unnecessarily large. In such a case, a receiver could + send a PING or other small ack-eliciting frame occasionally, such as + once per round trip, to elicit an ACK from the peer. + + In cases without ACK frame loss, this algorithm allows for a minimum + of 1 RTT of reordering. In cases with ACK frame loss and reordering, + this approach does not guarantee that every acknowledgment is seen by + the sender before it is no longer included in the ACK frame. Packets + could be received out of order, and all subsequent ACK frames + containing them could be lost. In this case, the loss recovery + algorithm could cause spurious retransmissions, but the sender will + continue making forward progress. + +13.2.5. Measuring and Reporting Host Delay + + An endpoint measures the delays intentionally introduced between the + time the packet with the largest packet number is received and the + time an acknowledgment is sent. The endpoint encodes this + acknowledgment delay in the ACK Delay field of an ACK frame; see + Section 19.3. This allows the receiver of the ACK frame to adjust + for any intentional delays, which is important for getting a better + estimate of the path RTT when acknowledgments are delayed. + + A packet might be held in the OS kernel or elsewhere on the host + before being processed. An endpoint MUST NOT include delays that it + does not control when populating the ACK Delay field in an ACK frame. + However, endpoints SHOULD include buffering delays caused by + unavailability of decryption keys, since these delays can be large + and are likely to be non-repeating. + + When the measured acknowledgment delay is larger than its + max_ack_delay, an endpoint SHOULD report the measured delay. This + information is especially useful during the handshake when delays + might be large; see Section 13.2.1. + +13.2.6. ACK Frames and Packet Protection + + ACK frames MUST only be carried in a packet that has the same packet + number space as the packet being acknowledged; see Section 12.1. For + instance, packets that are protected with 1-RTT keys MUST be + acknowledged in packets that are also protected with 1-RTT keys. + + Packets that a client sends with 0-RTT packet protection MUST be + acknowledged by the server in packets protected by 1-RTT keys. This + can mean that the client is unable to use these acknowledgments if + the server cryptographic handshake messages are delayed or lost. + Note that the same limitation applies to other data sent by the + server protected by the 1-RTT keys. + +13.2.7. PADDING Frames Consume Congestion Window + + Packets containing PADDING frames are considered to be in flight for + congestion control purposes [QUIC-RECOVERY]. Packets containing only + PADDING frames therefore consume congestion window but do not + generate acknowledgments that will open the congestion window. To + avoid a deadlock, a sender SHOULD ensure that other frames are sent + periodically in addition to PADDING frames to elicit acknowledgments + from the receiver. + +13.3. Retransmission of Information + + QUIC packets that are determined to be lost are not retransmitted + whole. The same applies to the frames that are contained within lost + packets. Instead, the information that might be carried in frames is + sent again in new frames as needed. + + New frames and packets are used to carry information that is + determined to have been lost. In general, information is sent again + when a packet containing that information is determined to be lost, + and sending ceases when a packet containing that information is + acknowledged. + + * Data sent in CRYPTO frames is retransmitted according to the rules + in [QUIC-RECOVERY], until all data has been acknowledged. Data in + CRYPTO frames for Initial and Handshake packets is discarded when + keys for the corresponding packet number space are discarded. + + * Application data sent in STREAM frames is retransmitted in new + STREAM frames unless the endpoint has sent a RESET_STREAM for that + stream. Once an endpoint sends a RESET_STREAM frame, no further + STREAM frames are needed. + + * ACK frames carry the most recent set of acknowledgments and the + acknowledgment delay from the largest acknowledged packet, as + described in Section 13.2.1. Delaying the transmission of packets + containing ACK frames or resending old ACK frames can cause the + peer to generate an inflated RTT sample or unnecessarily disable + ECN. + + * Cancellation of stream transmission, as carried in a RESET_STREAM + frame, is sent until acknowledged or until all stream data is + acknowledged by the peer (that is, either the "Reset Recvd" or + "Data Recvd" state is reached on the sending part of the stream). + The content of a RESET_STREAM frame MUST NOT change when it is + sent again. + + * Similarly, a request to cancel stream transmission, as encoded in + a STOP_SENDING frame, is sent until the receiving part of the + stream enters either a "Data Recvd" or "Reset Recvd" state; see + Section 3.5. + + * Connection close signals, including packets that contain + CONNECTION_CLOSE frames, are not sent again when packet loss is + detected. Resending these signals is described in Section 10. + + * The current connection maximum data is sent in MAX_DATA frames. + An updated value is sent in a MAX_DATA frame if the packet + containing the most recently sent MAX_DATA frame is declared lost + or when the endpoint decides to update the limit. Care is + necessary to avoid sending this frame too often, as the limit can + increase frequently and cause an unnecessarily large number of + MAX_DATA frames to be sent; see Section 4.2. + + * The current maximum stream data offset is sent in MAX_STREAM_DATA + frames. Like MAX_DATA, an updated value is sent when the packet + containing the most recent MAX_STREAM_DATA frame for a stream is + lost or when the limit is updated, with care taken to prevent the + frame from being sent too often. An endpoint SHOULD stop sending + MAX_STREAM_DATA frames when the receiving part of the stream + enters a "Size Known" or "Reset Recvd" state. + + * The limit on streams of a given type is sent in MAX_STREAMS + frames. Like MAX_DATA, an updated value is sent when a packet + containing the most recent MAX_STREAMS for a stream type frame is + declared lost or when the limit is updated, with care taken to + prevent the frame from being sent too often. + + * Blocked signals are carried in DATA_BLOCKED, STREAM_DATA_BLOCKED, + and STREAMS_BLOCKED frames. DATA_BLOCKED frames have connection + scope, STREAM_DATA_BLOCKED frames have stream scope, and + STREAMS_BLOCKED frames are scoped to a specific stream type. A + new frame is sent if a packet containing the most recent frame for + a scope is lost, but only while the endpoint is blocked on the + corresponding limit. These frames always include the limit that + is causing blocking at the time that they are transmitted. + + * A liveness or path validation check using PATH_CHALLENGE frames is + sent periodically until a matching PATH_RESPONSE frame is received + or until there is no remaining need for liveness or path + validation checking. PATH_CHALLENGE frames include a different + payload each time they are sent. + + * Responses to path validation using PATH_RESPONSE frames are sent + just once. The peer is expected to send more PATH_CHALLENGE + frames as necessary to evoke additional PATH_RESPONSE frames. + + * New connection IDs are sent in NEW_CONNECTION_ID frames and + retransmitted if the packet containing them is lost. + Retransmissions of this frame carry the same sequence number + value. Likewise, retired connection IDs are sent in + RETIRE_CONNECTION_ID frames and retransmitted if the packet + containing them is lost. + + * NEW_TOKEN frames are retransmitted if the packet containing them + is lost. No special support is made for detecting reordered and + duplicated NEW_TOKEN frames other than a direct comparison of the + frame contents. + + * PING and PADDING frames contain no information, so lost PING or + PADDING frames do not require repair. + + * The HANDSHAKE_DONE frame MUST be retransmitted until it is + acknowledged. + + Endpoints SHOULD prioritize retransmission of data over sending new + data, unless priorities specified by the application indicate + otherwise; see Section 2.3. + + Even though a sender is encouraged to assemble frames containing up- + to-date information every time it sends a packet, it is not forbidden + to retransmit copies of frames from lost packets. A sender that + retransmits copies of frames needs to handle decreases in available + payload size due to changes in packet number length, connection ID + length, and path MTU. A receiver MUST accept packets containing an + outdated frame, such as a MAX_DATA frame carrying a smaller maximum + data value than one found in an older packet. + + A sender SHOULD avoid retransmitting information from packets once + they are acknowledged. This includes packets that are acknowledged + after being declared lost, which can happen in the presence of + network reordering. Doing so requires senders to retain information + about packets after they are declared lost. A sender can discard + this information after a period of time elapses that adequately + allows for reordering, such as a PTO (Section 6.2 of + [QUIC-RECOVERY]), or based on other events, such as reaching a memory + limit. + + Upon detecting losses, a sender MUST take appropriate congestion + control action. The details of loss detection and congestion control + are described in [QUIC-RECOVERY]. + +13.4. Explicit Congestion Notification + + QUIC endpoints can use ECN [RFC3168] to detect and respond to network + congestion. ECN allows an endpoint to set an ECN-Capable Transport + (ECT) codepoint in the ECN field of an IP packet. A network node can + then indicate congestion by setting the ECN-CE codepoint in the ECN + field instead of dropping the packet [RFC8087]. Endpoints react to + reported congestion by reducing their sending rate in response, as + described in [QUIC-RECOVERY]. + + To enable ECN, a sending QUIC endpoint first determines whether a + path supports ECN marking and whether the peer reports the ECN values + in received IP headers; see Section 13.4.2. + +13.4.1. Reporting ECN Counts + + The use of ECN requires the receiving endpoint to read the ECN field + from an IP packet, which is not possible on all platforms. If an + endpoint does not implement ECN support or does not have access to + received ECN fields, it does not report ECN counts for packets it + receives. + + Even if an endpoint does not set an ECT field in packets it sends, + the endpoint MUST provide feedback about ECN markings it receives, if + these are accessible. Failing to report the ECN counts will cause + the sender to disable the use of ECN for this connection. + + On receiving an IP packet with an ECT(0), ECT(1), or ECN-CE + codepoint, an ECN-enabled endpoint accesses the ECN field and + increases the corresponding ECT(0), ECT(1), or ECN-CE count. These + ECN counts are included in subsequent ACK frames; see Sections 13.2 + and 19.3. + + Each packet number space maintains separate acknowledgment state and + separate ECN counts. Coalesced QUIC packets (see Section 12.2) share + the same IP header so the ECN counts are incremented once for each + coalesced QUIC packet. + + For example, if one each of an Initial, Handshake, and 1-RTT QUIC + packet are coalesced into a single UDP datagram, the ECN counts for + all three packet number spaces will be incremented by one each, based + on the ECN field of the single IP header. + + ECN counts are only incremented when QUIC packets from the received + IP packet are processed. As such, duplicate QUIC packets are not + processed and do not increase ECN counts; see Section 21.10 for + relevant security concerns. + +13.4.2. ECN Validation + + It is possible for faulty network devices to corrupt or erroneously + drop packets that carry a non-zero ECN codepoint. To ensure + connectivity in the presence of such devices, an endpoint validates + the ECN counts for each network path and disables the use of ECN on + that path if errors are detected. + + To perform ECN validation for a new path: + + * The endpoint sets an ECT(0) codepoint in the IP header of early + outgoing packets sent on a new path to the peer [RFC8311]. + + * The endpoint monitors whether all packets sent with an ECT + codepoint are eventually deemed lost (Section 6 of + [QUIC-RECOVERY]), indicating that ECN validation has failed. + + If an endpoint has cause to expect that IP packets with an ECT + codepoint might be dropped by a faulty network element, the endpoint + could set an ECT codepoint for only the first ten outgoing packets on + a path, or for a period of three PTOs (see Section 6.2 of + [QUIC-RECOVERY]). If all packets marked with non-zero ECN codepoints + are subsequently lost, it can disable marking on the assumption that + the marking caused the loss. + + An endpoint thus attempts to use ECN and validates this for each new + connection, when switching to a server's preferred address, and on + active connection migration to a new path. Appendix A.4 describes + one possible algorithm. + + Other methods of probing paths for ECN support are possible, as are + different marking strategies. Implementations MAY use other methods + defined in RFCs; see [RFC8311]. Implementations that use the ECT(1) + codepoint need to perform ECN validation using the reported ECT(1) + counts. + +13.4.2.1. Receiving ACK Frames with ECN Counts + + Erroneous application of ECN-CE markings by the network can result in + degraded connection performance. An endpoint that receives an ACK + frame with ECN counts therefore validates the counts before using + them. It performs this validation by comparing newly received counts + against those from the last successfully processed ACK frame. Any + increase in the ECN counts is validated based on the ECN markings + that were applied to packets that are newly acknowledged in the ACK + frame. + + If an ACK frame newly acknowledges a packet that the endpoint sent + with either the ECT(0) or ECT(1) codepoint set, ECN validation fails + if the corresponding ECN counts are not present in the ACK frame. + This check detects a network element that zeroes the ECN field or a + peer that does not report ECN markings. + + ECN validation also fails if the sum of the increase in ECT(0) and + ECN-CE counts is less than the number of newly acknowledged packets + that were originally sent with an ECT(0) marking. Similarly, ECN + validation fails if the sum of the increases to ECT(1) and ECN-CE + counts is less than the number of newly acknowledged packets sent + with an ECT(1) marking. These checks can detect remarking of ECN-CE + markings by the network. + + An endpoint could miss acknowledgments for a packet when ACK frames + are lost. It is therefore possible for the total increase in ECT(0), + ECT(1), and ECN-CE counts to be greater than the number of packets + that are newly acknowledged by an ACK frame. This is why ECN counts + are permitted to be larger than the total number of packets that are + acknowledged. + + Validating ECN counts from reordered ACK frames can result in + failure. An endpoint MUST NOT fail ECN validation as a result of + processing an ACK frame that does not increase the largest + acknowledged packet number. + + ECN validation can fail if the received total count for either ECT(0) + or ECT(1) exceeds the total number of packets sent with each + corresponding ECT codepoint. In particular, validation will fail + when an endpoint receives a non-zero ECN count corresponding to an + ECT codepoint that it never applied. This check detects when packets + are remarked to ECT(0) or ECT(1) in the network. + +13.4.2.2. ECN Validation Outcomes + + If validation fails, then the endpoint MUST disable ECN. It stops + setting the ECT codepoint in IP packets that it sends, assuming that + either the network path or the peer does not support ECN. + + Even if validation fails, an endpoint MAY revalidate ECN for the same + path at any later time in the connection. An endpoint could continue + to periodically attempt validation. + + Upon successful validation, an endpoint MAY continue to set an ECT + codepoint in subsequent packets it sends, with the expectation that + the path is ECN capable. Network routing and path elements can + change mid-connection; an endpoint MUST disable ECN if validation + later fails. + +14. Datagram Size + + A UDP datagram can include one or more QUIC packets. The datagram + size refers to the total UDP payload size of a single UDP datagram + carrying QUIC packets. The datagram size includes one or more QUIC + packet headers and protected payloads, but not the UDP or IP headers. + + The maximum datagram size is defined as the largest size of UDP + payload that can be sent across a network path using a single UDP + datagram. QUIC MUST NOT be used if the network path cannot support a + maximum datagram size of at least 1200 bytes. + + QUIC assumes a minimum IP packet size of at least 1280 bytes. This + is the IPv6 minimum size [IPv6] and is also supported by most modern + IPv4 networks. Assuming the minimum IP header size of 40 bytes for + IPv6 and 20 bytes for IPv4 and a UDP header size of 8 bytes, this + results in a maximum datagram size of 1232 bytes for IPv6 and 1252 + bytes for IPv4. Thus, modern IPv4 and all IPv6 network paths are + expected to be able to support QUIC. + + | Note: This requirement to support a UDP payload of 1200 bytes + | limits the space available for IPv6 extension headers to 32 + | bytes or IPv4 options to 52 bytes if the path only supports the + | IPv6 minimum MTU of 1280 bytes. This affects Initial packets + | and path validation. + + Any maximum datagram size larger than 1200 bytes can be discovered + using Path Maximum Transmission Unit Discovery (PMTUD) (see + Section 14.2.1) or Datagram Packetization Layer PMTU Discovery + (DPLPMTUD) (see Section 14.3). + + Enforcement of the max_udp_payload_size transport parameter + (Section 18.2) might act as an additional limit on the maximum + datagram size. A sender can avoid exceeding this limit, once the + value is known. However, prior to learning the value of the + transport parameter, endpoints risk datagrams being lost if they send + datagrams larger than the smallest allowed maximum datagram size of + 1200 bytes. + + UDP datagrams MUST NOT be fragmented at the IP layer. In IPv4 + [IPv4], the Don't Fragment (DF) bit MUST be set if possible, to + prevent fragmentation on the path. + + QUIC sometimes requires datagrams to be no smaller than a certain + size; see Section 8.1 as an example. However, the size of a datagram + is not authenticated. That is, if an endpoint receives a datagram of + a certain size, it cannot know that the sender sent the datagram at + the same size. Therefore, an endpoint MUST NOT close a connection + when it receives a datagram that does not meet size constraints; the + endpoint MAY discard such datagrams. + +14.1. Initial Datagram Size + + A client MUST expand the payload of all UDP datagrams carrying + Initial packets to at least the smallest allowed maximum datagram + size of 1200 bytes by adding PADDING frames to the Initial packet or + by coalescing the Initial packet; see Section 12.2. Initial packets + can even be coalesced with invalid packets, which a receiver will + discard. Similarly, a server MUST expand the payload of all UDP + datagrams carrying ack-eliciting Initial packets to at least the + smallest allowed maximum datagram size of 1200 bytes. + + Sending UDP datagrams of this size ensures that the network path + supports a reasonable Path Maximum Transmission Unit (PMTU), in both + directions. Additionally, a client that expands Initial packets + helps reduce the amplitude of amplification attacks caused by server + responses toward an unverified client address; see Section 8. + + Datagrams containing Initial packets MAY exceed 1200 bytes if the + sender believes that the network path and peer both support the size + that it chooses. + + A server MUST discard an Initial packet that is carried in a UDP + datagram with a payload that is smaller than the smallest allowed + maximum datagram size of 1200 bytes. A server MAY also immediately + close the connection by sending a CONNECTION_CLOSE frame with an + error code of PROTOCOL_VIOLATION; see Section 10.2.3. + + The server MUST also limit the number of bytes it sends before + validating the address of the client; see Section 8. + +14.2. Path Maximum Transmission Unit + + The PMTU is the maximum size of the entire IP packet, including the + IP header, UDP header, and UDP payload. The UDP payload includes one + or more QUIC packet headers and protected payloads. The PMTU can + depend on path characteristics and can therefore change over time. + The largest UDP payload an endpoint sends at any given time is + referred to as the endpoint's maximum datagram size. + + An endpoint SHOULD use DPLPMTUD (Section 14.3) or PMTUD + (Section 14.2.1) to determine whether the path to a destination will + support a desired maximum datagram size without fragmentation. In + the absence of these mechanisms, QUIC endpoints SHOULD NOT send + datagrams larger than the smallest allowed maximum datagram size. + + Both DPLPMTUD and PMTUD send datagrams that are larger than the + current maximum datagram size, referred to as PMTU probes. All QUIC + packets that are not sent in a PMTU probe SHOULD be sized to fit + within the maximum datagram size to avoid the datagram being + fragmented or dropped [RFC8085]. + + If a QUIC endpoint determines that the PMTU between any pair of local + and remote IP addresses cannot support the smallest allowed maximum + datagram size of 1200 bytes, it MUST immediately cease sending QUIC + packets, except for those in PMTU probes or those containing + CONNECTION_CLOSE frames, on the affected path. An endpoint MAY + terminate the connection if an alternative path cannot be found. + + Each pair of local and remote addresses could have a different PMTU. + QUIC implementations that implement any kind of PMTU discovery + therefore SHOULD maintain a maximum datagram size for each + combination of local and remote IP addresses. + + A QUIC implementation MAY be more conservative in computing the + maximum datagram size to allow for unknown tunnel overheads or IP + header options/extensions. + +14.2.1. Handling of ICMP Messages by PMTUD + + PMTUD [RFC1191] [RFC8201] relies on reception of ICMP messages (that + is, IPv6 Packet Too Big (PTB) messages) that indicate when an IP + packet is dropped because it is larger than the local router MTU. + DPLPMTUD can also optionally use these messages. This use of ICMP + messages is potentially vulnerable to attacks by entities that cannot + observe packets but might successfully guess the addresses used on + the path. These attacks could reduce the PMTU to a bandwidth- + inefficient value. + + An endpoint MUST ignore an ICMP message that claims the PMTU has + decreased below QUIC's smallest allowed maximum datagram size. + + The requirements for generating ICMP [RFC1812] [RFC4443] state that + the quoted packet should contain as much of the original packet as + possible without exceeding the minimum MTU for the IP version. The + size of the quoted packet can actually be smaller, or the information + unintelligible, as described in Section 1.1 of [DPLPMTUD]. + + QUIC endpoints using PMTUD SHOULD validate ICMP messages to protect + from packet injection as specified in [RFC8201] and Section 5.2 of + [RFC8085]. This validation SHOULD use the quoted packet supplied in + the payload of an ICMP message to associate the message with a + corresponding transport connection (see Section 4.6.1 of [DPLPMTUD]). + ICMP message validation MUST include matching IP addresses and UDP + ports [RFC8085] and, when possible, connection IDs to an active QUIC + session. The endpoint SHOULD ignore all ICMP messages that fail + validation. + + An endpoint MUST NOT increase the PMTU based on ICMP messages; see + Item 6 in Section 3 of [DPLPMTUD]. Any reduction in QUIC's maximum + datagram size in response to ICMP messages MAY be provisional until + QUIC's loss detection algorithm determines that the quoted packet has + actually been lost. + +14.3. Datagram Packetization Layer PMTU Discovery + + DPLPMTUD [DPLPMTUD] relies on tracking loss or acknowledgment of QUIC + packets that are carried in PMTU probes. PMTU probes for DPLPMTUD + that use the PADDING frame implement "Probing using padding data", as + defined in Section 4.1 of [DPLPMTUD]. + + Endpoints SHOULD set the initial value of BASE_PLPMTU (Section 5.1 of + [DPLPMTUD]) to be consistent with QUIC's smallest allowed maximum + datagram size. The MIN_PLPMTU is the same as the BASE_PLPMTU. + + QUIC endpoints implementing DPLPMTUD maintain a DPLPMTUD Maximum + Packet Size (MPS) (Section 4.4 of [DPLPMTUD]) for each combination of + local and remote IP addresses. This corresponds to the maximum + datagram size. + +14.3.1. DPLPMTUD and Initial Connectivity + + From the perspective of DPLPMTUD, QUIC is an acknowledged + Packetization Layer (PL). A QUIC sender can therefore enter the + DPLPMTUD BASE state (Section 5.2 of [DPLPMTUD]) when the QUIC + connection handshake has been completed. + +14.3.2. Validating the Network Path with DPLPMTUD + + QUIC is an acknowledged PL; therefore, a QUIC sender does not + implement a DPLPMTUD CONFIRMATION_TIMER while in the SEARCH_COMPLETE + state; see Section 5.2 of [DPLPMTUD]. + +14.3.3. Handling of ICMP Messages by DPLPMTUD + + An endpoint using DPLPMTUD requires the validation of any received + ICMP PTB message before using the PTB information, as defined in + Section 4.6 of [DPLPMTUD]. In addition to UDP port validation, QUIC + validates an ICMP message by using other PL information (e.g., + validation of connection IDs in the quoted packet of any received + ICMP message). + + The considerations for processing ICMP messages described in + Section 14.2.1 also apply if these messages are used by DPLPMTUD. + +14.4. Sending QUIC PMTU Probes + + PMTU probes are ack-eliciting packets. + + Endpoints could limit the content of PMTU probes to PING and PADDING + frames, since packets that are larger than the current maximum + datagram size are more likely to be dropped by the network. Loss of + a QUIC packet that is carried in a PMTU probe is therefore not a + reliable indication of congestion and SHOULD NOT trigger a congestion + control reaction; see Item 7 in Section 3 of [DPLPMTUD]. However, + PMTU probes consume congestion window, which could delay subsequent + transmission by an application. + +14.4.1. PMTU Probes Containing Source Connection ID + + Endpoints that rely on the Destination Connection ID field for + routing incoming QUIC packets are likely to require that the + connection ID be included in PMTU probes to route any resulting ICMP + messages (Section 14.2.1) back to the correct endpoint. However, + only long header packets (Section 17.2) contain the Source Connection + ID field, and long header packets are not decrypted or acknowledged + by the peer once the handshake is complete. + + One way to construct a PMTU probe is to coalesce (see Section 12.2) a + packet with a long header, such as a Handshake or 0-RTT packet + (Section 17.2), with a short header packet in a single UDP datagram. + If the resulting PMTU probe reaches the endpoint, the packet with the + long header will be ignored, but the short header packet will be + acknowledged. If the PMTU probe causes an ICMP message to be sent, + the first part of the probe will be quoted in that message. If the + Source Connection ID field is within the quoted portion of the probe, + that could be used for routing or validation of the ICMP message. + + | Note: The purpose of using a packet with a long header is only + | to ensure that the quoted packet contained in the ICMP message + | contains a Source Connection ID field. This packet does not + | need to be a valid packet, and it can be sent even if there is + | no current use for packets of that type. + +15. Versions + + QUIC versions are identified using a 32-bit unsigned number. + + The version 0x00000000 is reserved to represent version negotiation. + This version of the specification is identified by the number + 0x00000001. + + Other versions of QUIC might have different properties from this + version. The properties of QUIC that are guaranteed to be consistent + across all versions of the protocol are described in + [QUIC-INVARIANTS]. + + Version 0x00000001 of QUIC uses TLS as a cryptographic handshake + protocol, as described in [QUIC-TLS]. + + Versions with the most significant 16 bits of the version number + cleared are reserved for use in future IETF consensus documents. + + Versions that follow the pattern 0x?a?a?a?a are reserved for use in + forcing version negotiation to be exercised -- that is, any version + number where the low four bits of all bytes is 1010 (in binary). A + client or server MAY advertise support for any of these reserved + versions. + + Reserved version numbers will never represent a real protocol; a + client MAY use one of these version numbers with the expectation that + the server will initiate version negotiation; a server MAY advertise + support for one of these versions and can expect that clients ignore + the value. + +16. Variable-Length Integer Encoding + + QUIC packets and frames commonly use a variable-length encoding for + non-negative integer values. This encoding ensures that smaller + integer values need fewer bytes to encode. + + The QUIC variable-length integer encoding reserves the two most + significant bits of the first byte to encode the base-2 logarithm of + the integer encoding length in bytes. The integer value is encoded + on the remaining bits, in network byte order. + + This means that integers are encoded on 1, 2, 4, or 8 bytes and can + encode 6-, 14-, 30-, or 62-bit values, respectively. Table 4 + summarizes the encoding properties. + + +======+========+=============+=======================+ + | 2MSB | Length | Usable Bits | Range | + +======+========+=============+=======================+ + | 00 | 1 | 6 | 0-63 | + +------+--------+-------------+-----------------------+ + | 01 | 2 | 14 | 0-16383 | + +------+--------+-------------+-----------------------+ + | 10 | 4 | 30 | 0-1073741823 | + +------+--------+-------------+-----------------------+ + | 11 | 8 | 62 | 0-4611686018427387903 | + +------+--------+-------------+-----------------------+ + + Table 4: Summary of Integer Encodings + + An example of a decoding algorithm and sample encodings are shown in + Appendix A.1. + + Values do not need to be encoded on the minimum number of bytes + necessary, with the sole exception of the Frame Type field; see + Section 12.4. + + Versions (Section 15), packet numbers sent in the header + (Section 17.1), and the length of connection IDs in long header + packets (Section 17.2) are described using integers but do not use + this encoding. + +17. Packet Formats + + All numeric values are encoded in network byte order (that is, big + endian), and all field sizes are in bits. Hexadecimal notation is + used for describing the value of fields. + +17.1. Packet Number Encoding and Decoding + + Packet numbers are integers in the range 0 to 2^62-1 (Section 12.3). + When present in long or short packet headers, they are encoded in 1 + to 4 bytes. The number of bits required to represent the packet + number is reduced by including only the least significant bits of the + packet number. + + The encoded packet number is protected as described in Section 5.4 of + [QUIC-TLS]. + + Prior to receiving an acknowledgment for a packet number space, the + full packet number MUST be included; it is not to be truncated, as + described below. + + After an acknowledgment is received for a packet number space, the + sender MUST use a packet number size able to represent more than + twice as large a range as the difference between the largest + acknowledged packet number and the packet number being sent. A peer + receiving the packet will then correctly decode the packet number, + unless the packet is delayed in transit such that it arrives after + many higher-numbered packets have been received. An endpoint SHOULD + use a large enough packet number encoding to allow the packet number + to be recovered even if the packet arrives after packets that are + sent afterwards. + + As a result, the size of the packet number encoding is at least one + bit more than the base-2 logarithm of the number of contiguous + unacknowledged packet numbers, including the new packet. Pseudocode + and an example for packet number encoding can be found in + Appendix A.2. + + At a receiver, protection of the packet number is removed prior to + recovering the full packet number. The full packet number is then + reconstructed based on the number of significant bits present, the + value of those bits, and the largest packet number received in a + successfully authenticated packet. Recovering the full packet number + is necessary to successfully complete the removal of packet + protection. + + Once header protection is removed, the packet number is decoded by + finding the packet number value that is closest to the next expected + packet. The next expected packet is the highest received packet + number plus one. Pseudocode and an example for packet number + decoding can be found in Appendix A.3. + +17.2. Long Header Packets + + Long Header Packet { + Header Form (1) = 1, + Fixed Bit (1) = 1, + Long Packet Type (2), + Type-Specific Bits (4), + Version (32), + Destination Connection ID Length (8), + Destination Connection ID (0..160), + Source Connection ID Length (8), + Source Connection ID (0..160), + Type-Specific Payload (..), + } + + Figure 13: Long Header Packet Format + + Long headers are used for packets that are sent prior to the + establishment of 1-RTT keys. Once 1-RTT keys are available, a sender + switches to sending packets using the short header (Section 17.3). + The long form allows for special packets -- such as the Version + Negotiation packet -- to be represented in this uniform fixed-length + packet format. Packets that use the long header contain the + following fields: + + Header Form: The most significant bit (0x80) of byte 0 (the first + byte) is set to 1 for long headers. + + Fixed Bit: The next bit (0x40) of byte 0 is set to 1, unless the + packet is a Version Negotiation packet. Packets containing a zero + value for this bit are not valid packets in this version and MUST + be discarded. A value of 1 for this bit allows QUIC to coexist + with other protocols; see [RFC7983]. + + Long Packet Type: The next two bits (those with a mask of 0x30) of + byte 0 contain a packet type. Packet types are listed in Table 5. + + Type-Specific Bits: The semantics of the lower four bits (those with + a mask of 0x0f) of byte 0 are determined by the packet type. + + Version: The QUIC Version is a 32-bit field that follows the first + byte. This field indicates the version of QUIC that is in use and + determines how the rest of the protocol fields are interpreted. + + Destination Connection ID Length: The byte following the version + contains the length in bytes of the Destination Connection ID + field that follows it. This length is encoded as an 8-bit + unsigned integer. In QUIC version 1, this value MUST NOT exceed + 20 bytes. Endpoints that receive a version 1 long header with a + value larger than 20 MUST drop the packet. In order to properly + form a Version Negotiation packet, servers SHOULD be able to read + longer connection IDs from other QUIC versions. + + Destination Connection ID: The Destination Connection ID field + follows the Destination Connection ID Length field, which + indicates the length of this field. Section 7.2 describes the use + of this field in more detail. + + Source Connection ID Length: The byte following the Destination + Connection ID contains the length in bytes of the Source + Connection ID field that follows it. This length is encoded as an + 8-bit unsigned integer. In QUIC version 1, this value MUST NOT + exceed 20 bytes. Endpoints that receive a version 1 long header + with a value larger than 20 MUST drop the packet. In order to + properly form a Version Negotiation packet, servers SHOULD be able + to read longer connection IDs from other QUIC versions. + + Source Connection ID: The Source Connection ID field follows the + Source Connection ID Length field, which indicates the length of + this field. Section 7.2 describes the use of this field in more + detail. + + Type-Specific Payload: The remainder of the packet, if any, is type + specific. + + In this version of QUIC, the following packet types with the long + header are defined: + + +======+===========+================+ + | Type | Name | Section | + +======+===========+================+ + | 0x00 | Initial | Section 17.2.2 | + +------+-----------+----------------+ + | 0x01 | 0-RTT | Section 17.2.3 | + +------+-----------+----------------+ + | 0x02 | Handshake | Section 17.2.4 | + +------+-----------+----------------+ + | 0x03 | Retry | Section 17.2.5 | + +------+-----------+----------------+ + + Table 5: Long Header Packet Types + + The header form bit, Destination and Source Connection ID lengths, + Destination and Source Connection ID fields, and Version fields of a + long header packet are version independent. The other fields in the + first byte are version specific. See [QUIC-INVARIANTS] for details + on how packets from different versions of QUIC are interpreted. + + The interpretation of the fields and the payload are specific to a + version and packet type. While type-specific semantics for this + version are described in the following sections, several long header + packets in this version of QUIC contain these additional fields: + + Reserved Bits: Two bits (those with a mask of 0x0c) of byte 0 are + reserved across multiple packet types. These bits are protected + using header protection; see Section 5.4 of [QUIC-TLS]. The value + included prior to protection MUST be set to 0. An endpoint MUST + treat receipt of a packet that has a non-zero value for these bits + after removing both packet and header protection as a connection + error of type PROTOCOL_VIOLATION. Discarding such a packet after + only removing header protection can expose the endpoint to + attacks; see Section 9.5 of [QUIC-TLS]. + + Packet Number Length: In packet types that contain a Packet Number + field, the least significant two bits (those with a mask of 0x03) + of byte 0 contain the length of the Packet Number field, encoded + as an unsigned two-bit integer that is one less than the length of + the Packet Number field in bytes. That is, the length of the + Packet Number field is the value of this field plus one. These + bits are protected using header protection; see Section 5.4 of + [QUIC-TLS]. + + Length: This is the length of the remainder of the packet (that is, + the Packet Number and Payload fields) in bytes, encoded as a + variable-length integer (Section 16). + + Packet Number: This field is 1 to 4 bytes long. The packet number + is protected using header protection; see Section 5.4 of + [QUIC-TLS]. The length of the Packet Number field is encoded in + the Packet Number Length bits of byte 0; see above. + + Packet Payload: This is the payload of the packet -- containing a + sequence of frames -- that is protected using packet protection. + +17.2.1. Version Negotiation Packet + + A Version Negotiation packet is inherently not version specific. + Upon receipt by a client, it will be identified as a Version + Negotiation packet based on the Version field having a value of 0. + + The Version Negotiation packet is a response to a client packet that + contains a version that is not supported by the server. It is only + sent by servers. + + The layout of a Version Negotiation packet is: + + Version Negotiation Packet { + Header Form (1) = 1, + Unused (7), + Version (32) = 0, + Destination Connection ID Length (8), + Destination Connection ID (0..2040), + Source Connection ID Length (8), + Source Connection ID (0..2040), + Supported Version (32) ..., + } + + Figure 14: Version Negotiation Packet + + The value in the Unused field is set to an arbitrary value by the + server. Clients MUST ignore the value of this field. Where QUIC + might be multiplexed with other protocols (see [RFC7983]), servers + SHOULD set the most significant bit of this field (0x40) to 1 so that + Version Negotiation packets appear to have the Fixed Bit field. Note + that other versions of QUIC might not make a similar recommendation. + + The Version field of a Version Negotiation packet MUST be set to + 0x00000000. + + The server MUST include the value from the Source Connection ID field + of the packet it receives in the Destination Connection ID field. + The value for Source Connection ID MUST be copied from the + Destination Connection ID of the received packet, which is initially + randomly selected by a client. Echoing both connection IDs gives + clients some assurance that the server received the packet and that + the Version Negotiation packet was not generated by an entity that + did not observe the Initial packet. + + Future versions of QUIC could have different requirements for the + lengths of connection IDs. In particular, connection IDs might have + a smaller minimum length or a greater maximum length. Version- + specific rules for the connection ID therefore MUST NOT influence a + decision about whether to send a Version Negotiation packet. + + The remainder of the Version Negotiation packet is a list of 32-bit + versions that the server supports. + + A Version Negotiation packet is not acknowledged. It is only sent in + response to a packet that indicates an unsupported version; see + Section 5.2.2. + + The Version Negotiation packet does not include the Packet Number and + Length fields present in other packets that use the long header form. + Consequently, a Version Negotiation packet consumes an entire UDP + datagram. + + A server MUST NOT send more than one Version Negotiation packet in + response to a single UDP datagram. + + See Section 6 for a description of the version negotiation process. + +17.2.2. Initial Packet + + An Initial packet uses long headers with a type value of 0x00. It + carries the first CRYPTO frames sent by the client and server to + perform key exchange, and it carries ACK frames in either direction. + + Initial Packet { + Header Form (1) = 1, + Fixed Bit (1) = 1, + Long Packet Type (2) = 0, + Reserved Bits (2), + Packet Number Length (2), + Version (32), + Destination Connection ID Length (8), + Destination Connection ID (0..160), + Source Connection ID Length (8), + Source Connection ID (0..160), + Token Length (i), + Token (..), + Length (i), + Packet Number (8..32), + Packet Payload (8..), + } + + Figure 15: Initial Packet + + The Initial packet contains a long header as well as the Length and + Packet Number fields; see Section 17.2. The first byte contains the + Reserved and Packet Number Length bits; see also Section 17.2. + Between the Source Connection ID and Length fields, there are two + additional fields specific to the Initial packet. + + Token Length: A variable-length integer specifying the length of the + Token field, in bytes. This value is 0 if no token is present. + Initial packets sent by the server MUST set the Token Length field + to 0; clients that receive an Initial packet with a non-zero Token + Length field MUST either discard the packet or generate a + connection error of type PROTOCOL_VIOLATION. + + Token: The value of the token that was previously provided in a + Retry packet or NEW_TOKEN frame; see Section 8.1. + + In order to prevent tampering by version-unaware middleboxes, Initial + packets are protected with connection- and version-specific keys + (Initial keys) as described in [QUIC-TLS]. This protection does not + provide confidentiality or integrity against attackers that can + observe packets, but it does prevent attackers that cannot observe + packets from spoofing Initial packets. + + The client and server use the Initial packet type for any packet that + contains an initial cryptographic handshake message. This includes + all cases where a new packet containing the initial cryptographic + message needs to be created, such as the packets sent after receiving + a Retry packet; see Section 17.2.5. + + A server sends its first Initial packet in response to a client + Initial. A server MAY send multiple Initial packets. The + cryptographic key exchange could require multiple round trips or + retransmissions of this data. + + The payload of an Initial packet includes a CRYPTO frame (or frames) + containing a cryptographic handshake message, ACK frames, or both. + PING, PADDING, and CONNECTION_CLOSE frames of type 0x1c are also + permitted. An endpoint that receives an Initial packet containing + other frames can either discard the packet as spurious or treat it as + a connection error. + + The first packet sent by a client always includes a CRYPTO frame that + contains the start or all of the first cryptographic handshake + message. The first CRYPTO frame sent always begins at an offset of + 0; see Section 7. + + Note that if the server sends a TLS HelloRetryRequest (see + Section 4.7 of [QUIC-TLS]), the client will send another series of + Initial packets. These Initial packets will continue the + cryptographic handshake and will contain CRYPTO frames starting at an + offset matching the size of the CRYPTO frames sent in the first + flight of Initial packets. + +17.2.2.1. Abandoning Initial Packets + + A client stops both sending and processing Initial packets when it + sends its first Handshake packet. A server stops sending and + processing Initial packets when it receives its first Handshake + packet. Though packets might still be in flight or awaiting + acknowledgment, no further Initial packets need to be exchanged + beyond this point. Initial packet protection keys are discarded (see + Section 4.9.1 of [QUIC-TLS]) along with any loss recovery and + congestion control state; see Section 6.4 of [QUIC-RECOVERY]. + + Any data in CRYPTO frames is discarded -- and no longer retransmitted + -- when Initial keys are discarded. + +17.2.3. 0-RTT + + A 0-RTT packet uses long headers with a type value of 0x01, followed + by the Length and Packet Number fields; see Section 17.2. The first + byte contains the Reserved and Packet Number Length bits; see + Section 17.2. A 0-RTT packet is used to carry "early" data from the + client to the server as part of the first flight, prior to handshake + completion. As part of the TLS handshake, the server can accept or + reject this early data. + + See Section 2.3 of [TLS13] for a discussion of 0-RTT data and its + limitations. + + 0-RTT Packet { + Header Form (1) = 1, + Fixed Bit (1) = 1, + Long Packet Type (2) = 1, + Reserved Bits (2), + Packet Number Length (2), + Version (32), + Destination Connection ID Length (8), + Destination Connection ID (0..160), + Source Connection ID Length (8), + Source Connection ID (0..160), + Length (i), + Packet Number (8..32), + Packet Payload (8..), + } + + Figure 16: 0-RTT Packet + + Packet numbers for 0-RTT protected packets use the same space as + 1-RTT protected packets. + + After a client receives a Retry packet, 0-RTT packets are likely to + have been lost or discarded by the server. A client SHOULD attempt + to resend data in 0-RTT packets after it sends a new Initial packet. + New packet numbers MUST be used for any new packets that are sent; as + described in Section 17.2.5.3, reusing packet numbers could + compromise packet protection. + + A client only receives acknowledgments for its 0-RTT packets once the + handshake is complete, as defined in Section 4.1.1 of [QUIC-TLS]. + + A client MUST NOT send 0-RTT packets once it starts processing 1-RTT + packets from the server. This means that 0-RTT packets cannot + contain any response to frames from 1-RTT packets. For instance, a + client cannot send an ACK frame in a 0-RTT packet, because that can + only acknowledge a 1-RTT packet. An acknowledgment for a 1-RTT + packet MUST be carried in a 1-RTT packet. + + A server SHOULD treat a violation of remembered limits + (Section 7.4.1) as a connection error of an appropriate type (for + instance, a FLOW_CONTROL_ERROR for exceeding stream data limits). + +17.2.4. Handshake Packet + + A Handshake packet uses long headers with a type value of 0x02, + followed by the Length and Packet Number fields; see Section 17.2. + The first byte contains the Reserved and Packet Number Length bits; + see Section 17.2. It is used to carry cryptographic handshake + messages and acknowledgments from the server and client. + + Handshake Packet { + Header Form (1) = 1, + Fixed Bit (1) = 1, + Long Packet Type (2) = 2, + Reserved Bits (2), + Packet Number Length (2), + Version (32), + Destination Connection ID Length (8), + Destination Connection ID (0..160), + Source Connection ID Length (8), + Source Connection ID (0..160), + Length (i), + Packet Number (8..32), + Packet Payload (8..), + } + + Figure 17: Handshake Protected Packet + + Once a client has received a Handshake packet from a server, it uses + Handshake packets to send subsequent cryptographic handshake messages + and acknowledgments to the server. + + The Destination Connection ID field in a Handshake packet contains a + connection ID that is chosen by the recipient of the packet; the + Source Connection ID includes the connection ID that the sender of + the packet wishes to use; see Section 7.2. + + Handshake packets have their own packet number space, and thus the + first Handshake packet sent by a server contains a packet number of + 0. + + The payload of this packet contains CRYPTO frames and could contain + PING, PADDING, or ACK frames. Handshake packets MAY contain + CONNECTION_CLOSE frames of type 0x1c. Endpoints MUST treat receipt + of Handshake packets with other frames as a connection error of type + PROTOCOL_VIOLATION. + + Like Initial packets (see Section 17.2.2.1), data in CRYPTO frames + for Handshake packets is discarded -- and no longer retransmitted -- + when Handshake protection keys are discarded. + +17.2.5. Retry Packet + + As shown in Figure 18, a Retry packet uses a long packet header with + a type value of 0x03. It carries an address validation token created + by the server. It is used by a server that wishes to perform a + retry; see Section 8.1. + + Retry Packet { + Header Form (1) = 1, + Fixed Bit (1) = 1, + Long Packet Type (2) = 3, + Unused (4), + Version (32), + Destination Connection ID Length (8), + Destination Connection ID (0..160), + Source Connection ID Length (8), + Source Connection ID (0..160), + Retry Token (..), + Retry Integrity Tag (128), + } + + Figure 18: Retry Packet + + A Retry packet does not contain any protected fields. The value in + the Unused field is set to an arbitrary value by the server; a client + MUST ignore these bits. In addition to the fields from the long + header, it contains these additional fields: + + Retry Token: An opaque token that the server can use to validate the + client's address. + + Retry Integrity Tag: Defined in Section 5.8 ("Retry Packet + Integrity") of [QUIC-TLS]. + +17.2.5.1. Sending a Retry Packet + + The server populates the Destination Connection ID with the + connection ID that the client included in the Source Connection ID of + the Initial packet. + + The server includes a connection ID of its choice in the Source + Connection ID field. This value MUST NOT be equal to the Destination + Connection ID field of the packet sent by the client. A client MUST + discard a Retry packet that contains a Source Connection ID field + that is identical to the Destination Connection ID field of its + Initial packet. The client MUST use the value from the Source + Connection ID field of the Retry packet in the Destination Connection + ID field of subsequent packets that it sends. + + A server MAY send Retry packets in response to Initial and 0-RTT + packets. A server can either discard or buffer 0-RTT packets that it + receives. A server can send multiple Retry packets as it receives + Initial or 0-RTT packets. A server MUST NOT send more than one Retry + packet in response to a single UDP datagram. + +17.2.5.2. Handling a Retry Packet + + A client MUST accept and process at most one Retry packet for each + connection attempt. After the client has received and processed an + Initial or Retry packet from the server, it MUST discard any + subsequent Retry packets that it receives. + + Clients MUST discard Retry packets that have a Retry Integrity Tag + that cannot be validated; see Section 5.8 of [QUIC-TLS]. This + diminishes an attacker's ability to inject a Retry packet and + protects against accidental corruption of Retry packets. A client + MUST discard a Retry packet with a zero-length Retry Token field. + + The client responds to a Retry packet with an Initial packet that + includes the provided Retry token to continue connection + establishment. + + A client sets the Destination Connection ID field of this Initial + packet to the value from the Source Connection ID field in the Retry + packet. Changing the Destination Connection ID field also results in + a change to the keys used to protect the Initial packet. It also + sets the Token field to the token provided in the Retry packet. The + client MUST NOT change the Source Connection ID because the server + could include the connection ID as part of its token validation + logic; see Section 8.1.4. + + A Retry packet does not include a packet number and cannot be + explicitly acknowledged by a client. + +17.2.5.3. Continuing a Handshake after Retry + + Subsequent Initial packets from the client include the connection ID + and token values from the Retry packet. The client copies the Source + Connection ID field from the Retry packet to the Destination + Connection ID field and uses this value until an Initial packet with + an updated value is received; see Section 7.2. The value of the + Token field is copied to all subsequent Initial packets; see + Section 8.1.2. + + Other than updating the Destination Connection ID and Token fields, + the Initial packet sent by the client is subject to the same + restrictions as the first Initial packet. A client MUST use the same + cryptographic handshake message it included in this packet. A server + MAY treat a packet that contains a different cryptographic handshake + message as a connection error or discard it. Note that including a + Token field reduces the available space for the cryptographic + handshake message, which might result in the client needing to send + multiple Initial packets. + + A client MAY attempt 0-RTT after receiving a Retry packet by sending + 0-RTT packets to the connection ID provided by the server. + + A client MUST NOT reset the packet number for any packet number space + after processing a Retry packet. In particular, 0-RTT packets + contain confidential information that will most likely be + retransmitted on receiving a Retry packet. The keys used to protect + these new 0-RTT packets will not change as a result of responding to + a Retry packet. However, the data sent in these packets could be + different than what was sent earlier. Sending these new packets with + the same packet number is likely to compromise the packet protection + for those packets because the same key and nonce could be used to + protect different content. A server MAY abort the connection if it + detects that the client reset the packet number. + + The connection IDs used in Initial and Retry packets exchanged + between client and server are copied to the transport parameters and + validated as described in Section 7.3. + +17.3. Short Header Packets + + This version of QUIC defines a single packet type that uses the short + packet header. + +17.3.1. 1-RTT Packet + + A 1-RTT packet uses a short packet header. It is used after the + version and 1-RTT keys are negotiated. + + 1-RTT Packet { + Header Form (1) = 0, + Fixed Bit (1) = 1, + Spin Bit (1), + Reserved Bits (2), + Key Phase (1), + Packet Number Length (2), + Destination Connection ID (0..160), + Packet Number (8..32), + Packet Payload (8..), + } + + Figure 19: 1-RTT Packet + + 1-RTT packets contain the following fields: + + Header Form: The most significant bit (0x80) of byte 0 is set to 0 + for the short header. + + Fixed Bit: The next bit (0x40) of byte 0 is set to 1. Packets + containing a zero value for this bit are not valid packets in this + version and MUST be discarded. A value of 1 for this bit allows + QUIC to coexist with other protocols; see [RFC7983]. + + Spin Bit: The third most significant bit (0x20) of byte 0 is the + latency spin bit, set as described in Section 17.4. + + Reserved Bits: The next two bits (those with a mask of 0x18) of byte + 0 are reserved. These bits are protected using header protection; + see Section 5.4 of [QUIC-TLS]. The value included prior to + protection MUST be set to 0. An endpoint MUST treat receipt of a + packet that has a non-zero value for these bits, after removing + both packet and header protection, as a connection error of type + PROTOCOL_VIOLATION. Discarding such a packet after only removing + header protection can expose the endpoint to attacks; see + Section 9.5 of [QUIC-TLS]. + + Key Phase: The next bit (0x04) of byte 0 indicates the key phase, + which allows a recipient of a packet to identify the packet + protection keys that are used to protect the packet. See + [QUIC-TLS] for details. This bit is protected using header + protection; see Section 5.4 of [QUIC-TLS]. + + Packet Number Length: The least significant two bits (those with a + mask of 0x03) of byte 0 contain the length of the Packet Number + field, encoded as an unsigned two-bit integer that is one less + than the length of the Packet Number field in bytes. That is, the + length of the Packet Number field is the value of this field plus + one. These bits are protected using header protection; see + Section 5.4 of [QUIC-TLS]. + + Destination Connection ID: The Destination Connection ID is a + connection ID that is chosen by the intended recipient of the + packet. See Section 5.1 for more details. + + Packet Number: The Packet Number field is 1 to 4 bytes long. The + packet number is protected using header protection; see + Section 5.4 of [QUIC-TLS]. The length of the Packet Number field + is encoded in Packet Number Length field. See Section 17.1 for + details. + + Packet Payload: 1-RTT packets always include a 1-RTT protected + payload. + + The header form bit and the Destination Connection ID field of a + short header packet are version independent. The remaining fields + are specific to the selected QUIC version. See [QUIC-INVARIANTS] for + details on how packets from different versions of QUIC are + interpreted. + +17.4. Latency Spin Bit + + The latency spin bit, which is defined for 1-RTT packets + (Section 17.3.1), enables passive latency monitoring from observation + points on the network path throughout the duration of a connection. + The server reflects the spin value received, while the client "spins" + it after one RTT. On-path observers can measure the time between two + spin bit toggle events to estimate the end-to-end RTT of a + connection. + + The spin bit is only present in 1-RTT packets, since it is possible + to measure the initial RTT of a connection by observing the + handshake. Therefore, the spin bit is available after version + negotiation and connection establishment are completed. On-path + measurement and use of the latency spin bit are further discussed in + [QUIC-MANAGEABILITY]. + + The spin bit is an OPTIONAL feature of this version of QUIC. An + endpoint that does not support this feature MUST disable it, as + defined below. + + Each endpoint unilaterally decides if the spin bit is enabled or + disabled for a connection. Implementations MUST allow administrators + of clients and servers to disable the spin bit either globally or on + a per-connection basis. Even when the spin bit is not disabled by + the administrator, endpoints MUST disable their use of the spin bit + for a random selection of at least one in every 16 network paths, or + for one in every 16 connection IDs, in order to ensure that QUIC + connections that disable the spin bit are commonly observed on the + network. As each endpoint disables the spin bit independently, this + ensures that the spin bit signal is disabled on approximately one in + eight network paths. + + When the spin bit is disabled, endpoints MAY set the spin bit to any + value and MUST ignore any incoming value. It is RECOMMENDED that + endpoints set the spin bit to a random value either chosen + independently for each packet or chosen independently for each + connection ID. + + If the spin bit is enabled for the connection, the endpoint maintains + a spin value for each network path and sets the spin bit in the + packet header to the currently stored value when a 1-RTT packet is + sent on that path. The spin value is initialized to 0 in the + endpoint for each network path. Each endpoint also remembers the + highest packet number seen from its peer on each path. + + When a server receives a 1-RTT packet that increases the highest + packet number seen by the server from the client on a given network + path, it sets the spin value for that path to be equal to the spin + bit in the received packet. + + When a client receives a 1-RTT packet that increases the highest + packet number seen by the client from the server on a given network + path, it sets the spin value for that path to the inverse of the spin + bit in the received packet. + + An endpoint resets the spin value for a network path to 0 when + changing the connection ID being used on that network path. + +18. Transport Parameter Encoding + + The extension_data field of the quic_transport_parameters extension + defined in [QUIC-TLS] contains the QUIC transport parameters. They + are encoded as a sequence of transport parameters, as shown in + Figure 20: + + Transport Parameters { + Transport Parameter (..) ..., + } + + Figure 20: Sequence of Transport Parameters + + Each transport parameter is encoded as an (identifier, length, value) + tuple, as shown in Figure 21: + + Transport Parameter { + Transport Parameter ID (i), + Transport Parameter Length (i), + Transport Parameter Value (..), + } + + Figure 21: Transport Parameter Encoding + + The Transport Parameter Length field contains the length of the + Transport Parameter Value field in bytes. + + QUIC encodes transport parameters into a sequence of bytes, which is + then included in the cryptographic handshake. + +18.1. Reserved Transport Parameters + + Transport parameters with an identifier of the form "31 * N + 27" for + integer values of N are reserved to exercise the requirement that + unknown transport parameters be ignored. These transport parameters + have no semantics and can carry arbitrary values. + +18.2. Transport Parameter Definitions + + This section details the transport parameters defined in this + document. + + Many transport parameters listed here have integer values. Those + transport parameters that are identified as integers use a variable- + length integer encoding; see Section 16. Transport parameters have a + default value of 0 if the transport parameter is absent, unless + otherwise stated. + + The following transport parameters are defined: + + original_destination_connection_id (0x00): This parameter is the + value of the Destination Connection ID field from the first + Initial packet sent by the client; see Section 7.3. This + transport parameter is only sent by a server. + + max_idle_timeout (0x01): The maximum idle timeout is a value in + milliseconds that is encoded as an integer; see (Section 10.1). + Idle timeout is disabled when both endpoints omit this transport + parameter or specify a value of 0. + + stateless_reset_token (0x02): A stateless reset token is used in + verifying a stateless reset; see Section 10.3. This parameter is + a sequence of 16 bytes. This transport parameter MUST NOT be sent + by a client but MAY be sent by a server. A server that does not + send this transport parameter cannot use stateless reset + (Section 10.3) for the connection ID negotiated during the + handshake. + + max_udp_payload_size (0x03): The maximum UDP payload size parameter + is an integer value that limits the size of UDP payloads that the + endpoint is willing to receive. UDP datagrams with payloads + larger than this limit are not likely to be processed by the + receiver. + + The default for this parameter is the maximum permitted UDP + payload of 65527. Values below 1200 are invalid. + + This limit does act as an additional constraint on datagram size + in the same way as the path MTU, but it is a property of the + endpoint and not the path; see Section 14. It is expected that + this is the space an endpoint dedicates to holding incoming + packets. + + initial_max_data (0x04): The initial maximum data parameter is an + integer value that contains the initial value for the maximum + amount of data that can be sent on the connection. This is + equivalent to sending a MAX_DATA (Section 19.9) for the connection + immediately after completing the handshake. + + initial_max_stream_data_bidi_local (0x05): This parameter is an + integer value specifying the initial flow control limit for + locally initiated bidirectional streams. This limit applies to + newly created bidirectional streams opened by the endpoint that + sends the transport parameter. In client transport parameters, + this applies to streams with an identifier with the least + significant two bits set to 0x00; in server transport parameters, + this applies to streams with the least significant two bits set to + 0x01. + + initial_max_stream_data_bidi_remote (0x06): This parameter is an + integer value specifying the initial flow control limit for peer- + initiated bidirectional streams. This limit applies to newly + created bidirectional streams opened by the endpoint that receives + the transport parameter. In client transport parameters, this + applies to streams with an identifier with the least significant + two bits set to 0x01; in server transport parameters, this applies + to streams with the least significant two bits set to 0x00. + + initial_max_stream_data_uni (0x07): This parameter is an integer + value specifying the initial flow control limit for unidirectional + streams. This limit applies to newly created unidirectional + streams opened by the endpoint that receives the transport + parameter. In client transport parameters, this applies to + streams with an identifier with the least significant two bits set + to 0x03; in server transport parameters, this applies to streams + with the least significant two bits set to 0x02. + + initial_max_streams_bidi (0x08): The initial maximum bidirectional + streams parameter is an integer value that contains the initial + maximum number of bidirectional streams the endpoint that receives + this transport parameter is permitted to initiate. If this + parameter is absent or zero, the peer cannot open bidirectional + streams until a MAX_STREAMS frame is sent. Setting this parameter + is equivalent to sending a MAX_STREAMS (Section 19.11) of the + corresponding type with the same value. + + initial_max_streams_uni (0x09): The initial maximum unidirectional + streams parameter is an integer value that contains the initial + maximum number of unidirectional streams the endpoint that + receives this transport parameter is permitted to initiate. If + this parameter is absent or zero, the peer cannot open + unidirectional streams until a MAX_STREAMS frame is sent. Setting + this parameter is equivalent to sending a MAX_STREAMS + (Section 19.11) of the corresponding type with the same value. + + ack_delay_exponent (0x0a): The acknowledgment delay exponent is an + integer value indicating an exponent used to decode the ACK Delay + field in the ACK frame (Section 19.3). If this value is absent, a + default value of 3 is assumed (indicating a multiplier of 8). + Values above 20 are invalid. + + max_ack_delay (0x0b): The maximum acknowledgment delay is an integer + value indicating the maximum amount of time in milliseconds by + which the endpoint will delay sending acknowledgments. This value + SHOULD include the receiver's expected delays in alarms firing. + For example, if a receiver sets a timer for 5ms and alarms + commonly fire up to 1ms late, then it should send a max_ack_delay + of 6ms. If this value is absent, a default of 25 milliseconds is + assumed. Values of 2^14 or greater are invalid. + + disable_active_migration (0x0c): The disable active migration + transport parameter is included if the endpoint does not support + active connection migration (Section 9) on the address being used + during the handshake. An endpoint that receives this transport + parameter MUST NOT use a new local address when sending to the + address that the peer used during the handshake. This transport + parameter does not prohibit connection migration after a client + has acted on a preferred_address transport parameter. This + parameter is a zero-length value. + + preferred_address (0x0d): The server's preferred address is used to + effect a change in server address at the end of the handshake, as + described in Section 9.6. This transport parameter is only sent + by a server. Servers MAY choose to only send a preferred address + of one address family by sending an all-zero address and port + (0.0.0.0:0 or [::]:0) for the other family. IP addresses are + encoded in network byte order. + + The preferred_address transport parameter contains an address and + port for both IPv4 and IPv6. The four-byte IPv4 Address field is + followed by the associated two-byte IPv4 Port field. This is + followed by a 16-byte IPv6 Address field and two-byte IPv6 Port + field. After address and port pairs, a Connection ID Length field + describes the length of the following Connection ID field. + Finally, a 16-byte Stateless Reset Token field includes the + stateless reset token associated with the connection ID. The + format of this transport parameter is shown in Figure 22 below. + + The Connection ID field and the Stateless Reset Token field + contain an alternative connection ID that has a sequence number of + 1; see Section 5.1.1. Having these values sent alongside the + preferred address ensures that there will be at least one unused + active connection ID when the client initiates migration to the + preferred address. + + The Connection ID and Stateless Reset Token fields of a preferred + address are identical in syntax and semantics to the corresponding + fields of a NEW_CONNECTION_ID frame (Section 19.15). A server + that chooses a zero-length connection ID MUST NOT provide a + preferred address. Similarly, a server MUST NOT include a zero- + length connection ID in this transport parameter. A client MUST + treat a violation of these requirements as a connection error of + type TRANSPORT_PARAMETER_ERROR. + + Preferred Address { + IPv4 Address (32), + IPv4 Port (16), + IPv6 Address (128), + IPv6 Port (16), + Connection ID Length (8), + Connection ID (..), + Stateless Reset Token (128), + } + + Figure 22: Preferred Address Format + + active_connection_id_limit (0x0e): This is an integer value + specifying the maximum number of connection IDs from the peer that + an endpoint is willing to store. This value includes the + connection ID received during the handshake, that received in the + preferred_address transport parameter, and those received in + NEW_CONNECTION_ID frames. The value of the + active_connection_id_limit parameter MUST be at least 2. An + endpoint that receives a value less than 2 MUST close the + connection with an error of type TRANSPORT_PARAMETER_ERROR. If + this transport parameter is absent, a default of 2 is assumed. If + an endpoint issues a zero-length connection ID, it will never send + a NEW_CONNECTION_ID frame and therefore ignores the + active_connection_id_limit value received from its peer. + + initial_source_connection_id (0x0f): This is the value that the + endpoint included in the Source Connection ID field of the first + Initial packet it sends for the connection; see Section 7.3. + + retry_source_connection_id (0x10): This is the value that the server + included in the Source Connection ID field of a Retry packet; see + Section 7.3. This transport parameter is only sent by a server. + + If present, transport parameters that set initial per-stream flow + control limits (initial_max_stream_data_bidi_local, + initial_max_stream_data_bidi_remote, and initial_max_stream_data_uni) + are equivalent to sending a MAX_STREAM_DATA frame (Section 19.10) on + every stream of the corresponding type immediately after opening. If + the transport parameter is absent, streams of that type start with a + flow control limit of 0. + + A client MUST NOT include any server-only transport parameter: + original_destination_connection_id, preferred_address, + retry_source_connection_id, or stateless_reset_token. A server MUST + treat receipt of any of these transport parameters as a connection + error of type TRANSPORT_PARAMETER_ERROR. + +19. Frame Types and Formats + + As described in Section 12.4, packets contain one or more frames. + This section describes the format and semantics of the core QUIC + frame types. + +19.1. PADDING Frames + + A PADDING frame (type=0x00) has no semantic value. PADDING frames + can be used to increase the size of a packet. Padding can be used to + increase an Initial packet to the minimum required size or to provide + protection against traffic analysis for protected packets. + + PADDING frames are formatted as shown in Figure 23, which shows that + PADDING frames have no content. That is, a PADDING frame consists of + the single byte that identifies the frame as a PADDING frame. + + PADDING Frame { + Type (i) = 0x00, + } + + Figure 23: PADDING Frame Format + +19.2. PING Frames + + Endpoints can use PING frames (type=0x01) to verify that their peers + are still alive or to check reachability to the peer. + + PING frames are formatted as shown in Figure 24, which shows that + PING frames have no content. + + PING Frame { + Type (i) = 0x01, + } + + Figure 24: PING Frame Format + + The receiver of a PING frame simply needs to acknowledge the packet + containing this frame. + + The PING frame can be used to keep a connection alive when an + application or application protocol wishes to prevent the connection + from timing out; see Section 10.1.2. + +19.3. ACK Frames + + Receivers send ACK frames (types 0x02 and 0x03) to inform senders of + packets they have received and processed. The ACK frame contains one + or more ACK Ranges. ACK Ranges identify acknowledged packets. If + the frame type is 0x03, ACK frames also contain the cumulative count + of QUIC packets with associated ECN marks received on the connection + up until this point. QUIC implementations MUST properly handle both + types, and, if they have enabled ECN for packets they send, they + SHOULD use the information in the ECN section to manage their + congestion state. + + QUIC acknowledgments are irrevocable. Once acknowledged, a packet + remains acknowledged, even if it does not appear in a future ACK + frame. This is unlike reneging for TCP Selective Acknowledgments + (SACKs) [RFC2018]. + + Packets from different packet number spaces can be identified using + the same numeric value. An acknowledgment for a packet needs to + indicate both a packet number and a packet number space. This is + accomplished by having each ACK frame only acknowledge packet numbers + in the same space as the packet in which the ACK frame is contained. + + Version Negotiation and Retry packets cannot be acknowledged because + they do not contain a packet number. Rather than relying on ACK + frames, these packets are implicitly acknowledged by the next Initial + packet sent by the client. + + ACK frames are formatted as shown in Figure 25. + + ACK Frame { + Type (i) = 0x02..0x03, + Largest Acknowledged (i), + ACK Delay (i), + ACK Range Count (i), + First ACK Range (i), + ACK Range (..) ..., + [ECN Counts (..)], + } + + Figure 25: ACK Frame Format + + ACK frames contain the following fields: + + Largest Acknowledged: A variable-length integer representing the + largest packet number the peer is acknowledging; this is usually + the largest packet number that the peer has received prior to + generating the ACK frame. Unlike the packet number in the QUIC + long or short header, the value in an ACK frame is not truncated. + + ACK Delay: A variable-length integer encoding the acknowledgment + delay in microseconds; see Section 13.2.5. It is decoded by + multiplying the value in the field by 2 to the power of the + ack_delay_exponent transport parameter sent by the sender of the + ACK frame; see Section 18.2. Compared to simply expressing the + delay as an integer, this encoding allows for a larger range of + values within the same number of bytes, at the cost of lower + resolution. + + ACK Range Count: A variable-length integer specifying the number of + ACK Range fields in the frame. + + First ACK Range: A variable-length integer indicating the number of + contiguous packets preceding the Largest Acknowledged that are + being acknowledged. That is, the smallest packet acknowledged in + the range is determined by subtracting the First ACK Range value + from the Largest Acknowledged field. + + ACK Ranges: Contains additional ranges of packets that are + alternately not acknowledged (Gap) and acknowledged (ACK Range); + see Section 19.3.1. + + ECN Counts: The three ECN counts; see Section 19.3.2. + +19.3.1. ACK Ranges + + Each ACK Range consists of alternating Gap and ACK Range Length + values in descending packet number order. ACK Ranges can be + repeated. The number of Gap and ACK Range Length values is + determined by the ACK Range Count field; one of each value is present + for each value in the ACK Range Count field. + + ACK Ranges are structured as shown in Figure 26. + + ACK Range { + Gap (i), + ACK Range Length (i), + } + + Figure 26: ACK Ranges + + The fields that form each ACK Range are: + + Gap: A variable-length integer indicating the number of contiguous + unacknowledged packets preceding the packet number one lower than + the smallest in the preceding ACK Range. + + ACK Range Length: A variable-length integer indicating the number of + contiguous acknowledged packets preceding the largest packet + number, as determined by the preceding Gap. + + Gap and ACK Range Length values use a relative integer encoding for + efficiency. Though each encoded value is positive, the values are + subtracted, so that each ACK Range describes progressively lower- + numbered packets. + + Each ACK Range acknowledges a contiguous range of packets by + indicating the number of acknowledged packets that precede the + largest packet number in that range. A value of 0 indicates that + only the largest packet number is acknowledged. Larger ACK Range + values indicate a larger range, with corresponding lower values for + the smallest packet number in the range. Thus, given a largest + packet number for the range, the smallest value is determined by the + following formula: + + smallest = largest - ack_range + + An ACK Range acknowledges all packets between the smallest packet + number and the largest, inclusive. + + The largest value for an ACK Range is determined by cumulatively + subtracting the size of all preceding ACK Range Lengths and Gaps. + + Each Gap indicates a range of packets that are not being + acknowledged. The number of packets in the gap is one higher than + the encoded value of the Gap field. + + The value of the Gap field establishes the largest packet number + value for the subsequent ACK Range using the following formula: + + largest = previous_smallest - gap - 2 + + If any computed packet number is negative, an endpoint MUST generate + a connection error of type FRAME_ENCODING_ERROR. + +19.3.2. ECN Counts + + The ACK frame uses the least significant bit of the type value (that + is, type 0x03) to indicate ECN feedback and report receipt of QUIC + packets with associated ECN codepoints of ECT(0), ECT(1), or ECN-CE + in the packet's IP header. ECN counts are only present when the ACK + frame type is 0x03. + + When present, there are three ECN counts, as shown in Figure 27. + + ECN Counts { + ECT0 Count (i), + ECT1 Count (i), + ECN-CE Count (i), + } + + Figure 27: ECN Count Format + + The ECN count fields are: + + ECT0 Count: A variable-length integer representing the total number + of packets received with the ECT(0) codepoint in the packet number + space of the ACK frame. + + ECT1 Count: A variable-length integer representing the total number + of packets received with the ECT(1) codepoint in the packet number + space of the ACK frame. + + ECN-CE Count: A variable-length integer representing the total + number of packets received with the ECN-CE codepoint in the packet + number space of the ACK frame. + + ECN counts are maintained separately for each packet number space. + +19.4. RESET_STREAM Frames + + An endpoint uses a RESET_STREAM frame (type=0x04) to abruptly + terminate the sending part of a stream. + + After sending a RESET_STREAM, an endpoint ceases transmission and + retransmission of STREAM frames on the identified stream. A receiver + of RESET_STREAM can discard any data that it already received on that + stream. + + An endpoint that receives a RESET_STREAM frame for a send-only stream + MUST terminate the connection with error STREAM_STATE_ERROR. + + RESET_STREAM frames are formatted as shown in Figure 28. + + RESET_STREAM Frame { + Type (i) = 0x04, + Stream ID (i), + Application Protocol Error Code (i), + Final Size (i), + } + + Figure 28: RESET_STREAM Frame Format + + RESET_STREAM frames contain the following fields: + + Stream ID: A variable-length integer encoding of the stream ID of + the stream being terminated. + + Application Protocol Error Code: A variable-length integer + containing the application protocol error code (see Section 20.2) + that indicates why the stream is being closed. + + Final Size: A variable-length integer indicating the final size of + the stream by the RESET_STREAM sender, in units of bytes; see + Section 4.5. + +19.5. STOP_SENDING Frames + + An endpoint uses a STOP_SENDING frame (type=0x05) to communicate that + incoming data is being discarded on receipt per application request. + STOP_SENDING requests that a peer cease transmission on a stream. + + A STOP_SENDING frame can be sent for streams in the "Recv" or "Size + Known" states; see Section 3.2. Receiving a STOP_SENDING frame for a + locally initiated stream that has not yet been created MUST be + treated as a connection error of type STREAM_STATE_ERROR. An + endpoint that receives a STOP_SENDING frame for a receive-only stream + MUST terminate the connection with error STREAM_STATE_ERROR. + + STOP_SENDING frames are formatted as shown in Figure 29. + + STOP_SENDING Frame { + Type (i) = 0x05, + Stream ID (i), + Application Protocol Error Code (i), + } + + Figure 29: STOP_SENDING Frame Format + + STOP_SENDING frames contain the following fields: + + Stream ID: A variable-length integer carrying the stream ID of the + stream being ignored. + + Application Protocol Error Code: A variable-length integer + containing the application-specified reason the sender is ignoring + the stream; see Section 20.2. + +19.6. CRYPTO Frames + + A CRYPTO frame (type=0x06) is used to transmit cryptographic + handshake messages. It can be sent in all packet types except 0-RTT. + The CRYPTO frame offers the cryptographic protocol an in-order stream + of bytes. CRYPTO frames are functionally identical to STREAM frames, + except that they do not bear a stream identifier; they are not flow + controlled; and they do not carry markers for optional offset, + optional length, and the end of the stream. + + CRYPTO frames are formatted as shown in Figure 30. + + CRYPTO Frame { + Type (i) = 0x06, + Offset (i), + Length (i), + Crypto Data (..), + } + + Figure 30: CRYPTO Frame Format + + CRYPTO frames contain the following fields: + + Offset: A variable-length integer specifying the byte offset in the + stream for the data in this CRYPTO frame. + + Length: A variable-length integer specifying the length of the + Crypto Data field in this CRYPTO frame. + + Crypto Data: The cryptographic message data. + + There is a separate flow of cryptographic handshake data in each + encryption level, each of which starts at an offset of 0. This + implies that each encryption level is treated as a separate CRYPTO + stream of data. + + The largest offset delivered on a stream -- the sum of the offset and + data length -- cannot exceed 2^62-1. Receipt of a frame that exceeds + this limit MUST be treated as a connection error of type + FRAME_ENCODING_ERROR or CRYPTO_BUFFER_EXCEEDED. + + Unlike STREAM frames, which include a stream ID indicating to which + stream the data belongs, the CRYPTO frame carries data for a single + stream per encryption level. The stream does not have an explicit + end, so CRYPTO frames do not have a FIN bit. + +19.7. NEW_TOKEN Frames + + A server sends a NEW_TOKEN frame (type=0x07) to provide the client + with a token to send in the header of an Initial packet for a future + connection. + + NEW_TOKEN frames are formatted as shown in Figure 31. + + NEW_TOKEN Frame { + Type (i) = 0x07, + Token Length (i), + Token (..), + } + + Figure 31: NEW_TOKEN Frame Format + + NEW_TOKEN frames contain the following fields: + + Token Length: A variable-length integer specifying the length of the + token in bytes. + + Token: An opaque blob that the client can use with a future Initial + packet. The token MUST NOT be empty. A client MUST treat receipt + of a NEW_TOKEN frame with an empty Token field as a connection + error of type FRAME_ENCODING_ERROR. + + A client might receive multiple NEW_TOKEN frames that contain the + same token value if packets containing the frame are incorrectly + determined to be lost. Clients are responsible for discarding + duplicate values, which might be used to link connection attempts; + see Section 8.1.3. + + Clients MUST NOT send NEW_TOKEN frames. A server MUST treat receipt + of a NEW_TOKEN frame as a connection error of type + PROTOCOL_VIOLATION. + +19.8. STREAM Frames + + STREAM frames implicitly create a stream and carry stream data. The + Type field in the STREAM frame takes the form 0b00001XXX (or the set + of values from 0x08 to 0x0f). The three low-order bits of the frame + type determine the fields that are present in the frame: + + * The OFF bit (0x04) in the frame type is set to indicate that there + is an Offset field present. When set to 1, the Offset field is + present. When set to 0, the Offset field is absent and the Stream + Data starts at an offset of 0 (that is, the frame contains the + first bytes of the stream, or the end of a stream that includes no + data). + + * The LEN bit (0x02) in the frame type is set to indicate that there + is a Length field present. If this bit is set to 0, the Length + field is absent and the Stream Data field extends to the end of + the packet. If this bit is set to 1, the Length field is present. + + * The FIN bit (0x01) indicates that the frame marks the end of the + stream. The final size of the stream is the sum of the offset and + the length of this frame. + + An endpoint MUST terminate the connection with error + STREAM_STATE_ERROR if it receives a STREAM frame for a locally + initiated stream that has not yet been created, or for a send-only + stream. + + STREAM frames are formatted as shown in Figure 32. + + STREAM Frame { + Type (i) = 0x08..0x0f, + Stream ID (i), + [Offset (i)], + [Length (i)], + Stream Data (..), + } + + Figure 32: STREAM Frame Format + + STREAM frames contain the following fields: + + Stream ID: A variable-length integer indicating the stream ID of the + stream; see Section 2.1. + + Offset: A variable-length integer specifying the byte offset in the + stream for the data in this STREAM frame. This field is present + when the OFF bit is set to 1. When the Offset field is absent, + the offset is 0. + + Length: A variable-length integer specifying the length of the + Stream Data field in this STREAM frame. This field is present + when the LEN bit is set to 1. When the LEN bit is set to 0, the + Stream Data field consumes all the remaining bytes in the packet. + + Stream Data: The bytes from the designated stream to be delivered. + + When a Stream Data field has a length of 0, the offset in the STREAM + frame is the offset of the next byte that would be sent. + + The first byte in the stream has an offset of 0. The largest offset + delivered on a stream -- the sum of the offset and data length -- + cannot exceed 2^62-1, as it is not possible to provide flow control + credit for that data. Receipt of a frame that exceeds this limit + MUST be treated as a connection error of type FRAME_ENCODING_ERROR or + FLOW_CONTROL_ERROR. + +19.9. MAX_DATA Frames + + A MAX_DATA frame (type=0x10) is used in flow control to inform the + peer of the maximum amount of data that can be sent on the connection + as a whole. + + MAX_DATA frames are formatted as shown in Figure 33. + + MAX_DATA Frame { + Type (i) = 0x10, + Maximum Data (i), + } + + Figure 33: MAX_DATA Frame Format + + MAX_DATA frames contain the following field: + + Maximum Data: A variable-length integer indicating the maximum + amount of data that can be sent on the entire connection, in units + of bytes. + + All data sent in STREAM frames counts toward this limit. The sum of + the final sizes on all streams -- including streams in terminal + states -- MUST NOT exceed the value advertised by a receiver. An + endpoint MUST terminate a connection with an error of type + FLOW_CONTROL_ERROR if it receives more data than the maximum data + value that it has sent. This includes violations of remembered + limits in Early Data; see Section 7.4.1. + +19.10. MAX_STREAM_DATA Frames + + A MAX_STREAM_DATA frame (type=0x11) is used in flow control to inform + a peer of the maximum amount of data that can be sent on a stream. + + A MAX_STREAM_DATA frame can be sent for streams in the "Recv" state; + see Section 3.2. Receiving a MAX_STREAM_DATA frame for a locally + initiated stream that has not yet been created MUST be treated as a + connection error of type STREAM_STATE_ERROR. An endpoint that + receives a MAX_STREAM_DATA frame for a receive-only stream MUST + terminate the connection with error STREAM_STATE_ERROR. + + MAX_STREAM_DATA frames are formatted as shown in Figure 34. + + MAX_STREAM_DATA Frame { + Type (i) = 0x11, + Stream ID (i), + Maximum Stream Data (i), + } + + Figure 34: MAX_STREAM_DATA Frame Format + + MAX_STREAM_DATA frames contain the following fields: + + Stream ID: The stream ID of the affected stream, encoded as a + variable-length integer. + + Maximum Stream Data: A variable-length integer indicating the + maximum amount of data that can be sent on the identified stream, + in units of bytes. + + When counting data toward this limit, an endpoint accounts for the + largest received offset of data that is sent or received on the + stream. Loss or reordering can mean that the largest received offset + on a stream can be greater than the total size of data received on + that stream. Receiving STREAM frames might not increase the largest + received offset. + + The data sent on a stream MUST NOT exceed the largest maximum stream + data value advertised by the receiver. An endpoint MUST terminate a + connection with an error of type FLOW_CONTROL_ERROR if it receives + more data than the largest maximum stream data that it has sent for + the affected stream. This includes violations of remembered limits + in Early Data; see Section 7.4.1. + +19.11. MAX_STREAMS Frames + + A MAX_STREAMS frame (type=0x12 or 0x13) informs the peer of the + cumulative number of streams of a given type it is permitted to open. + A MAX_STREAMS frame with a type of 0x12 applies to bidirectional + streams, and a MAX_STREAMS frame with a type of 0x13 applies to + unidirectional streams. + + MAX_STREAMS frames are formatted as shown in Figure 35. + + MAX_STREAMS Frame { + Type (i) = 0x12..0x13, + Maximum Streams (i), + } + + Figure 35: MAX_STREAMS Frame Format + + MAX_STREAMS frames contain the following field: + + Maximum Streams: A count of the cumulative number of streams of the + corresponding type that can be opened over the lifetime of the + connection. This value cannot exceed 2^60, as it is not possible + to encode stream IDs larger than 2^62-1. Receipt of a frame that + permits opening of a stream larger than this limit MUST be treated + as a connection error of type FRAME_ENCODING_ERROR. + + Loss or reordering can cause an endpoint to receive a MAX_STREAMS + frame with a lower stream limit than was previously received. + MAX_STREAMS frames that do not increase the stream limit MUST be + ignored. + + An endpoint MUST NOT open more streams than permitted by the current + stream limit set by its peer. For instance, a server that receives a + unidirectional stream limit of 3 is permitted to open streams 3, 7, + and 11, but not stream 15. An endpoint MUST terminate a connection + with an error of type STREAM_LIMIT_ERROR if a peer opens more streams + than was permitted. This includes violations of remembered limits in + Early Data; see Section 7.4.1. + + Note that these frames (and the corresponding transport parameters) + do not describe the number of streams that can be opened + concurrently. The limit includes streams that have been closed as + well as those that are open. + +19.12. DATA_BLOCKED Frames + + A sender SHOULD send a DATA_BLOCKED frame (type=0x14) when it wishes + to send data but is unable to do so due to connection-level flow + control; see Section 4. DATA_BLOCKED frames can be used as input to + tuning of flow control algorithms; see Section 4.2. + + DATA_BLOCKED frames are formatted as shown in Figure 36. + + DATA_BLOCKED Frame { + Type (i) = 0x14, + Maximum Data (i), + } + + Figure 36: DATA_BLOCKED Frame Format + + DATA_BLOCKED frames contain the following field: + + Maximum Data: A variable-length integer indicating the connection- + level limit at which blocking occurred. + +19.13. STREAM_DATA_BLOCKED Frames + + A sender SHOULD send a STREAM_DATA_BLOCKED frame (type=0x15) when it + wishes to send data but is unable to do so due to stream-level flow + control. This frame is analogous to DATA_BLOCKED (Section 19.12). + + An endpoint that receives a STREAM_DATA_BLOCKED frame for a send-only + stream MUST terminate the connection with error STREAM_STATE_ERROR. + + STREAM_DATA_BLOCKED frames are formatted as shown in Figure 37. + + STREAM_DATA_BLOCKED Frame { + Type (i) = 0x15, + Stream ID (i), + Maximum Stream Data (i), + } + + Figure 37: STREAM_DATA_BLOCKED Frame Format + + STREAM_DATA_BLOCKED frames contain the following fields: + + Stream ID: A variable-length integer indicating the stream that is + blocked due to flow control. + + Maximum Stream Data: A variable-length integer indicating the offset + of the stream at which the blocking occurred. + +19.14. STREAMS_BLOCKED Frames + + A sender SHOULD send a STREAMS_BLOCKED frame (type=0x16 or 0x17) when + it wishes to open a stream but is unable to do so due to the maximum + stream limit set by its peer; see Section 19.11. A STREAMS_BLOCKED + frame of type 0x16 is used to indicate reaching the bidirectional + stream limit, and a STREAMS_BLOCKED frame of type 0x17 is used to + indicate reaching the unidirectional stream limit. + + A STREAMS_BLOCKED frame does not open the stream, but informs the + peer that a new stream was needed and the stream limit prevented the + creation of the stream. + + STREAMS_BLOCKED frames are formatted as shown in Figure 38. + + STREAMS_BLOCKED Frame { + Type (i) = 0x16..0x17, + Maximum Streams (i), + } + + Figure 38: STREAMS_BLOCKED Frame Format + + STREAMS_BLOCKED frames contain the following field: + + Maximum Streams: A variable-length integer indicating the maximum + number of streams allowed at the time the frame was sent. This + value cannot exceed 2^60, as it is not possible to encode stream + IDs larger than 2^62-1. Receipt of a frame that encodes a larger + stream ID MUST be treated as a connection error of type + STREAM_LIMIT_ERROR or FRAME_ENCODING_ERROR. + +19.15. NEW_CONNECTION_ID Frames + + An endpoint sends a NEW_CONNECTION_ID frame (type=0x18) to provide + its peer with alternative connection IDs that can be used to break + linkability when migrating connections; see Section 9.5. + + NEW_CONNECTION_ID frames are formatted as shown in Figure 39. + + NEW_CONNECTION_ID Frame { + Type (i) = 0x18, + Sequence Number (i), + Retire Prior To (i), + Length (8), + Connection ID (8..160), + Stateless Reset Token (128), + } + + Figure 39: NEW_CONNECTION_ID Frame Format + + NEW_CONNECTION_ID frames contain the following fields: + + Sequence Number: The sequence number assigned to the connection ID + by the sender, encoded as a variable-length integer; see + Section 5.1.1. + + Retire Prior To: A variable-length integer indicating which + connection IDs should be retired; see Section 5.1.2. + + Length: An 8-bit unsigned integer containing the length of the + connection ID. Values less than 1 and greater than 20 are invalid + and MUST be treated as a connection error of type + FRAME_ENCODING_ERROR. + + Connection ID: A connection ID of the specified length. + + Stateless Reset Token: A 128-bit value that will be used for a + stateless reset when the associated connection ID is used; see + Section 10.3. + + An endpoint MUST NOT send this frame if it currently requires that + its peer send packets with a zero-length Destination Connection ID. + Changing the length of a connection ID to or from zero length makes + it difficult to identify when the value of the connection ID changed. + An endpoint that is sending packets with a zero-length Destination + Connection ID MUST treat receipt of a NEW_CONNECTION_ID frame as a + connection error of type PROTOCOL_VIOLATION. + + Transmission errors, timeouts, and retransmissions might cause the + same NEW_CONNECTION_ID frame to be received multiple times. Receipt + of the same frame multiple times MUST NOT be treated as a connection + error. A receiver can use the sequence number supplied in the + NEW_CONNECTION_ID frame to handle receiving the same + NEW_CONNECTION_ID frame multiple times. + + If an endpoint receives a NEW_CONNECTION_ID frame that repeats a + previously issued connection ID with a different Stateless Reset + Token field value or a different Sequence Number field value, or if a + sequence number is used for different connection IDs, the endpoint + MAY treat that receipt as a connection error of type + PROTOCOL_VIOLATION. + + The Retire Prior To field applies to connection IDs established + during connection setup and the preferred_address transport + parameter; see Section 5.1.2. The value in the Retire Prior To field + MUST be less than or equal to the value in the Sequence Number field. + Receiving a value in the Retire Prior To field that is greater than + that in the Sequence Number field MUST be treated as a connection + error of type FRAME_ENCODING_ERROR. + + Once a sender indicates a Retire Prior To value, smaller values sent + in subsequent NEW_CONNECTION_ID frames have no effect. A receiver + MUST ignore any Retire Prior To fields that do not increase the + largest received Retire Prior To value. + + An endpoint that receives a NEW_CONNECTION_ID frame with a sequence + number smaller than the Retire Prior To field of a previously + received NEW_CONNECTION_ID frame MUST send a corresponding + RETIRE_CONNECTION_ID frame that retires the newly received connection + ID, unless it has already done so for that sequence number. + +19.16. RETIRE_CONNECTION_ID Frames + + An endpoint sends a RETIRE_CONNECTION_ID frame (type=0x19) to + indicate that it will no longer use a connection ID that was issued + by its peer. This includes the connection ID provided during the + handshake. Sending a RETIRE_CONNECTION_ID frame also serves as a + request to the peer to send additional connection IDs for future use; + see Section 5.1. New connection IDs can be delivered to a peer using + the NEW_CONNECTION_ID frame (Section 19.15). + + Retiring a connection ID invalidates the stateless reset token + associated with that connection ID. + + RETIRE_CONNECTION_ID frames are formatted as shown in Figure 40. + + RETIRE_CONNECTION_ID Frame { + Type (i) = 0x19, + Sequence Number (i), + } + + Figure 40: RETIRE_CONNECTION_ID Frame Format + + RETIRE_CONNECTION_ID frames contain the following field: + + Sequence Number: The sequence number of the connection ID being + retired; see Section 5.1.2. + + Receipt of a RETIRE_CONNECTION_ID frame containing a sequence number + greater than any previously sent to the peer MUST be treated as a + connection error of type PROTOCOL_VIOLATION. + + The sequence number specified in a RETIRE_CONNECTION_ID frame MUST + NOT refer to the Destination Connection ID field of the packet in + which the frame is contained. The peer MAY treat this as a + connection error of type PROTOCOL_VIOLATION. + + An endpoint cannot send this frame if it was provided with a zero- + length connection ID by its peer. An endpoint that provides a zero- + length connection ID MUST treat receipt of a RETIRE_CONNECTION_ID + frame as a connection error of type PROTOCOL_VIOLATION. + +19.17. PATH_CHALLENGE Frames + + Endpoints can use PATH_CHALLENGE frames (type=0x1a) to check + reachability to the peer and for path validation during connection + migration. + + PATH_CHALLENGE frames are formatted as shown in Figure 41. + + PATH_CHALLENGE Frame { + Type (i) = 0x1a, + Data (64), + } + + Figure 41: PATH_CHALLENGE Frame Format + + PATH_CHALLENGE frames contain the following field: + + Data: This 8-byte field contains arbitrary data. + + Including 64 bits of entropy in a PATH_CHALLENGE frame ensures that + it is easier to receive the packet than it is to guess the value + correctly. + + The recipient of this frame MUST generate a PATH_RESPONSE frame + (Section 19.18) containing the same Data value. + +19.18. PATH_RESPONSE Frames + + A PATH_RESPONSE frame (type=0x1b) is sent in response to a + PATH_CHALLENGE frame. + + PATH_RESPONSE frames are formatted as shown in Figure 42. The format + of a PATH_RESPONSE frame is identical to that of the PATH_CHALLENGE + frame; see Section 19.17. + + PATH_RESPONSE Frame { + Type (i) = 0x1b, + Data (64), + } + + Figure 42: PATH_RESPONSE Frame Format + + If the content of a PATH_RESPONSE frame does not match the content of + a PATH_CHALLENGE frame previously sent by the endpoint, the endpoint + MAY generate a connection error of type PROTOCOL_VIOLATION. + +19.19. CONNECTION_CLOSE Frames + + An endpoint sends a CONNECTION_CLOSE frame (type=0x1c or 0x1d) to + notify its peer that the connection is being closed. The + CONNECTION_CLOSE frame with a type of 0x1c is used to signal errors + at only the QUIC layer, or the absence of errors (with the NO_ERROR + code). The CONNECTION_CLOSE frame with a type of 0x1d is used to + signal an error with the application that uses QUIC. + + If there are open streams that have not been explicitly closed, they + are implicitly closed when the connection is closed. + + CONNECTION_CLOSE frames are formatted as shown in Figure 43. + + CONNECTION_CLOSE Frame { + Type (i) = 0x1c..0x1d, + Error Code (i), + [Frame Type (i)], + Reason Phrase Length (i), + Reason Phrase (..), + } + + Figure 43: CONNECTION_CLOSE Frame Format + + CONNECTION_CLOSE frames contain the following fields: + + Error Code: A variable-length integer that indicates the reason for + closing this connection. A CONNECTION_CLOSE frame of type 0x1c + uses codes from the space defined in Section 20.1. A + CONNECTION_CLOSE frame of type 0x1d uses codes defined by the + application protocol; see Section 20.2. + + Frame Type: A variable-length integer encoding the type of frame + that triggered the error. A value of 0 (equivalent to the mention + of the PADDING frame) is used when the frame type is unknown. The + application-specific variant of CONNECTION_CLOSE (type 0x1d) does + not include this field. + + Reason Phrase Length: A variable-length integer specifying the + length of the reason phrase in bytes. Because a CONNECTION_CLOSE + frame cannot be split between packets, any limits on packet size + will also limit the space available for a reason phrase. + + Reason Phrase: Additional diagnostic information for the closure. + This can be zero length if the sender chooses not to give details + beyond the Error Code value. This SHOULD be a UTF-8 encoded + string [RFC3629], though the frame does not carry information, + such as language tags, that would aid comprehension by any entity + other than the one that created the text. + + The application-specific variant of CONNECTION_CLOSE (type 0x1d) can + only be sent using 0-RTT or 1-RTT packets; see Section 12.5. When an + application wishes to abandon a connection during the handshake, an + endpoint can send a CONNECTION_CLOSE frame (type 0x1c) with an error + code of APPLICATION_ERROR in an Initial or Handshake packet. + +19.20. HANDSHAKE_DONE Frames + + The server uses a HANDSHAKE_DONE frame (type=0x1e) to signal + confirmation of the handshake to the client. + + HANDSHAKE_DONE frames are formatted as shown in Figure 44, which + shows that HANDSHAKE_DONE frames have no content. + + HANDSHAKE_DONE Frame { + Type (i) = 0x1e, + } + + Figure 44: HANDSHAKE_DONE Frame Format + + A HANDSHAKE_DONE frame can only be sent by the server. Servers MUST + NOT send a HANDSHAKE_DONE frame before completing the handshake. A + server MUST treat receipt of a HANDSHAKE_DONE frame as a connection + error of type PROTOCOL_VIOLATION. + +19.21. Extension Frames + + QUIC frames do not use a self-describing encoding. An endpoint + therefore needs to understand the syntax of all frames before it can + successfully process a packet. This allows for efficient encoding of + frames, but it means that an endpoint cannot send a frame of a type + that is unknown to its peer. + + An extension to QUIC that wishes to use a new type of frame MUST + first ensure that a peer is able to understand the frame. An + endpoint can use a transport parameter to signal its willingness to + receive extension frame types. One transport parameter can indicate + support for one or more extension frame types. + + Extensions that modify or replace core protocol functionality + (including frame types) will be difficult to combine with other + extensions that modify or replace the same functionality unless the + behavior of the combination is explicitly defined. Such extensions + SHOULD define their interaction with previously defined extensions + modifying the same protocol components. + + Extension frames MUST be congestion controlled and MUST cause an ACK + frame to be sent. The exception is extension frames that replace or + supplement the ACK frame. Extension frames are not included in flow + control unless specified in the extension. + + An IANA registry is used to manage the assignment of frame types; see + Section 22.4. + +20. Error Codes + + QUIC transport error codes and application error codes are 62-bit + unsigned integers. + +20.1. Transport Error Codes + + This section lists the defined QUIC transport error codes that can be + used in a CONNECTION_CLOSE frame with a type of 0x1c. These errors + apply to the entire connection. + + NO_ERROR (0x00): An endpoint uses this with CONNECTION_CLOSE to + signal that the connection is being closed abruptly in the absence + of any error. + + INTERNAL_ERROR (0x01): The endpoint encountered an internal error + and cannot continue with the connection. + + CONNECTION_REFUSED (0x02): The server refused to accept a new + connection. + + FLOW_CONTROL_ERROR (0x03): An endpoint received more data than it + permitted in its advertised data limits; see Section 4. + + STREAM_LIMIT_ERROR (0x04): An endpoint received a frame for a stream + identifier that exceeded its advertised stream limit for the + corresponding stream type. + + STREAM_STATE_ERROR (0x05): An endpoint received a frame for a stream + that was not in a state that permitted that frame; see Section 3. + + FINAL_SIZE_ERROR (0x06): (1) An endpoint received a STREAM frame + containing data that exceeded the previously established final + size, (2) an endpoint received a STREAM frame or a RESET_STREAM + frame containing a final size that was lower than the size of + stream data that was already received, or (3) an endpoint received + a STREAM frame or a RESET_STREAM frame containing a different + final size to the one already established. + + FRAME_ENCODING_ERROR (0x07): An endpoint received a frame that was + badly formatted -- for instance, a frame of an unknown type or an + ACK frame that has more acknowledgment ranges than the remainder + of the packet could carry. + + TRANSPORT_PARAMETER_ERROR (0x08): An endpoint received transport + parameters that were badly formatted, included an invalid value, + omitted a mandatory transport parameter, included a forbidden + transport parameter, or were otherwise in error. + + CONNECTION_ID_LIMIT_ERROR (0x09): The number of connection IDs + provided by the peer exceeds the advertised + active_connection_id_limit. + + PROTOCOL_VIOLATION (0x0a): An endpoint detected an error with + protocol compliance that was not covered by more specific error + codes. + + INVALID_TOKEN (0x0b): A server received a client Initial that + contained an invalid Token field. + + APPLICATION_ERROR (0x0c): The application or application protocol + caused the connection to be closed. + + CRYPTO_BUFFER_EXCEEDED (0x0d): An endpoint has received more data in + CRYPTO frames than it can buffer. + + KEY_UPDATE_ERROR (0x0e): An endpoint detected errors in performing + key updates; see Section 6 of [QUIC-TLS]. + + AEAD_LIMIT_REACHED (0x0f): An endpoint has reached the + confidentiality or integrity limit for the AEAD algorithm used by + the given connection. + + NO_VIABLE_PATH (0x10): An endpoint has determined that the network + path is incapable of supporting QUIC. An endpoint is unlikely to + receive a CONNECTION_CLOSE frame carrying this code except when + the path does not support a large enough MTU. + + CRYPTO_ERROR (0x0100-0x01ff): The cryptographic handshake failed. A + range of 256 values is reserved for carrying error codes specific + to the cryptographic handshake that is used. Codes for errors + occurring when TLS is used for the cryptographic handshake are + described in Section 4.8 of [QUIC-TLS]. + + See Section 22.5 for details on registering new error codes. + + In defining these error codes, several principles are applied. Error + conditions that might require specific action on the part of a + recipient are given unique codes. Errors that represent common + conditions are given specific codes. Absent either of these + conditions, error codes are used to identify a general function of + the stack, like flow control or transport parameter handling. + Finally, generic errors are provided for conditions where + implementations are unable or unwilling to use more specific codes. + +20.2. Application Protocol Error Codes + + The management of application error codes is left to application + protocols. Application protocol error codes are used for the + RESET_STREAM frame (Section 19.4), the STOP_SENDING frame + (Section 19.5), and the CONNECTION_CLOSE frame with a type of 0x1d + (Section 19.19). + +21. Security Considerations + + The goal of QUIC is to provide a secure transport connection. + Section 21.1 provides an overview of those properties; subsequent + sections discuss constraints and caveats regarding these properties, + including descriptions of known attacks and countermeasures. + +21.1. Overview of Security Properties + + A complete security analysis of QUIC is outside the scope of this + document. This section provides an informal description of the + desired security properties as an aid to implementers and to help + guide protocol analysis. + + QUIC assumes the threat model described in [SEC-CONS] and provides + protections against many of the attacks that arise from that model. + + For this purpose, attacks are divided into passive and active + attacks. Passive attackers have the ability to read packets from the + network, while active attackers also have the ability to write + packets into the network. However, a passive attack could involve an + attacker with the ability to cause a routing change or other + modification in the path taken by packets that comprise a connection. + + Attackers are additionally categorized as either on-path attackers or + off-path attackers. An on-path attacker can read, modify, or remove + any packet it observes such that the packet no longer reaches its + destination, while an off-path attacker observes the packets but + cannot prevent the original packet from reaching its intended + destination. Both types of attackers can also transmit arbitrary + packets. This definition differs from that of Section 3.5 of + [SEC-CONS] in that an off-path attacker is able to observe packets. + + Properties of the handshake, protected packets, and connection + migration are considered separately. + +21.1.1. Handshake + + The QUIC handshake incorporates the TLS 1.3 handshake and inherits + the cryptographic properties described in Appendix E.1 of [TLS13]. + Many of the security properties of QUIC depend on the TLS handshake + providing these properties. Any attack on the TLS handshake could + affect QUIC. + + Any attack on the TLS handshake that compromises the secrecy or + uniqueness of session keys, or the authentication of the + participating peers, affects other security guarantees provided by + QUIC that depend on those keys. For instance, migration (Section 9) + depends on the efficacy of confidentiality protections, both for the + negotiation of keys using the TLS handshake and for QUIC packet + protection, to avoid linkability across network paths. + + An attack on the integrity of the TLS handshake might allow an + attacker to affect the selection of application protocol or QUIC + version. + + In addition to the properties provided by TLS, the QUIC handshake + provides some defense against DoS attacks on the handshake. + +21.1.1.1. Anti-Amplification + + Address validation (Section 8) is used to verify that an entity that + claims a given address is able to receive packets at that address. + Address validation limits amplification attack targets to addresses + for which an attacker can observe packets. + + Prior to address validation, endpoints are limited in what they are + able to send. Endpoints cannot send data toward an unvalidated + address in excess of three times the data received from that address. + + | Note: The anti-amplification limit only applies when an + | endpoint responds to packets received from an unvalidated + | address. The anti-amplification limit does not apply to + | clients when establishing a new connection or when initiating + | connection migration. + +21.1.1.2. Server-Side DoS + + Computing the server's first flight for a full handshake is + potentially expensive, requiring both a signature and a key exchange + computation. In order to prevent computational DoS attacks, the + Retry packet provides a cheap token exchange mechanism that allows + servers to validate a client's IP address prior to doing any + expensive computations at the cost of a single round trip. After a + successful handshake, servers can issue new tokens to a client, which + will allow new connection establishment without incurring this cost. + +21.1.1.3. On-Path Handshake Termination + + An on-path or off-path attacker can force a handshake to fail by + replacing or racing Initial packets. Once valid Initial packets have + been exchanged, subsequent Handshake packets are protected with the + Handshake keys, and an on-path attacker cannot force handshake + failure other than by dropping packets to cause endpoints to abandon + the attempt. + + An on-path attacker can also replace the addresses of packets on + either side and therefore cause the client or server to have an + incorrect view of the remote addresses. Such an attack is + indistinguishable from the functions performed by a NAT. + +21.1.1.4. Parameter Negotiation + + The entire handshake is cryptographically protected, with the Initial + packets being encrypted with per-version keys and the Handshake and + later packets being encrypted with keys derived from the TLS key + exchange. Further, parameter negotiation is folded into the TLS + transcript and thus provides the same integrity guarantees as + ordinary TLS negotiation. An attacker can observe the client's + transport parameters (as long as it knows the version-specific salt) + but cannot observe the server's transport parameters and cannot + influence parameter negotiation. + + Connection IDs are unencrypted but integrity protected in all + packets. + + This version of QUIC does not incorporate a version negotiation + mechanism; implementations of incompatible versions will simply fail + to establish a connection. + +21.1.2. Protected Packets + + Packet protection (Section 12.1) applies authenticated encryption to + all packets except Version Negotiation packets, though Initial and + Retry packets have limited protection due to the use of version- + specific keying material; see [QUIC-TLS] for more details. This + section considers passive and active attacks against protected + packets. + + Both on-path and off-path attackers can mount a passive attack in + which they save observed packets for an offline attack against packet + protection at a future time; this is true for any observer of any + packet on any network. + + An attacker that injects packets without being able to observe valid + packets for a connection is unlikely to be successful, since packet + protection ensures that valid packets are only generated by endpoints + that possess the key material established during the handshake; see + Sections 7 and 21.1.1. Similarly, any active attacker that observes + packets and attempts to insert new data or modify existing data in + those packets should not be able to generate packets deemed valid by + the receiving endpoint, other than Initial packets. + + A spoofing attack, in which an active attacker rewrites unprotected + parts of a packet that it forwards or injects, such as the source or + destination address, is only effective if the attacker can forward + packets to the original endpoint. Packet protection ensures that the + packet payloads can only be processed by the endpoints that completed + the handshake, and invalid packets are ignored by those endpoints. + + An attacker can also modify the boundaries between packets and UDP + datagrams, causing multiple packets to be coalesced into a single + datagram or splitting coalesced packets into multiple datagrams. + Aside from datagrams containing Initial packets, which require + padding, modification of how packets are arranged in datagrams has no + functional effect on a connection, although it might change some + performance characteristics. + +21.1.3. Connection Migration + + Connection migration (Section 9) provides endpoints with the ability + to transition between IP addresses and ports on multiple paths, using + one path at a time for transmission and receipt of non-probing + frames. Path validation (Section 8.2) establishes that a peer is + both willing and able to receive packets sent on a particular path. + This helps reduce the effects of address spoofing by limiting the + number of packets sent to a spoofed address. + + This section describes the intended security properties of connection + migration under various types of DoS attacks. + +21.1.3.1. On-Path Active Attacks + + An attacker that can cause a packet it observes to no longer reach + its intended destination is considered an on-path attacker. When an + attacker is present between a client and server, endpoints are + required to send packets through the attacker to establish + connectivity on a given path. + + An on-path attacker can: + + * Inspect packets + + * Modify IP and UDP packet headers + + * Inject new packets + + * Delay packets + + * Reorder packets + + * Drop packets + + * Split and merge datagrams along packet boundaries + + An on-path attacker cannot: + + * Modify an authenticated portion of a packet and cause the + recipient to accept that packet + + An on-path attacker has the opportunity to modify the packets that it + observes; however, any modifications to an authenticated portion of a + packet will cause it to be dropped by the receiving endpoint as + invalid, as packet payloads are both authenticated and encrypted. + + QUIC aims to constrain the capabilities of an on-path attacker as + follows: + + 1. An on-path attacker can prevent the use of a path for a + connection, causing the connection to fail if it cannot use a + different path that does not contain the attacker. This can be + achieved by dropping all packets, modifying them so that they + fail to decrypt, or other methods. + + 2. An on-path attacker can prevent migration to a new path for which + the attacker is also on-path by causing path validation to fail + on the new path. + + 3. An on-path attacker cannot prevent a client from migrating to a + path for which the attacker is not on-path. + + 4. An on-path attacker can reduce the throughput of a connection by + delaying packets or dropping them. + + 5. An on-path attacker cannot cause an endpoint to accept a packet + for which it has modified an authenticated portion of that + packet. + +21.1.3.2. Off-Path Active Attacks + + An off-path attacker is not directly on the path between a client and + server but could be able to obtain copies of some or all packets sent + between the client and the server. It is also able to send copies of + those packets to either endpoint. + + An off-path attacker can: + + * Inspect packets + + * Inject new packets + + * Reorder injected packets + + An off-path attacker cannot: + + * Modify packets sent by endpoints + + * Delay packets + + * Drop packets + + * Reorder original packets + + An off-path attacker can create modified copies of packets that it + has observed and inject those copies into the network, potentially + with spoofed source and destination addresses. + + For the purposes of this discussion, it is assumed that an off-path + attacker has the ability to inject a modified copy of a packet into + the network that will reach the destination endpoint prior to the + arrival of the original packet observed by the attacker. In other + words, an attacker has the ability to consistently "win" a race with + the legitimate packets between the endpoints, potentially causing the + original packet to be ignored by the recipient. + + It is also assumed that an attacker has the resources necessary to + affect NAT state. In particular, an attacker can cause an endpoint + to lose its NAT binding and then obtain the same port for use with + its own traffic. + + QUIC aims to constrain the capabilities of an off-path attacker as + follows: + + 1. An off-path attacker can race packets and attempt to become a + "limited" on-path attacker. + + 2. An off-path attacker can cause path validation to succeed for + forwarded packets with the source address listed as the off-path + attacker as long as it can provide improved connectivity between + the client and the server. + + 3. An off-path attacker cannot cause a connection to close once the + handshake has completed. + + 4. An off-path attacker cannot cause migration to a new path to fail + if it cannot observe the new path. + + 5. An off-path attacker can become a limited on-path attacker during + migration to a new path for which it is also an off-path + attacker. + + 6. An off-path attacker can become a limited on-path attacker by + affecting shared NAT state such that it sends packets to the + server from the same IP address and port that the client + originally used. + +21.1.3.3. Limited On-Path Active Attacks + + A limited on-path attacker is an off-path attacker that has offered + improved routing of packets by duplicating and forwarding original + packets between the server and the client, causing those packets to + arrive before the original copies such that the original packets are + dropped by the destination endpoint. + + A limited on-path attacker differs from an on-path attacker in that + it is not on the original path between endpoints, and therefore the + original packets sent by an endpoint are still reaching their + destination. This means that a future failure to route copied + packets to the destination faster than their original path will not + prevent the original packets from reaching the destination. + + A limited on-path attacker can: + + * Inspect packets + + * Inject new packets + + * Modify unencrypted packet headers + + * Reorder packets + + A limited on-path attacker cannot: + + * Delay packets so that they arrive later than packets sent on the + original path + + * Drop packets + + * Modify the authenticated and encrypted portion of a packet and + cause the recipient to accept that packet + + A limited on-path attacker can only delay packets up to the point + that the original packets arrive before the duplicate packets, + meaning that it cannot offer routing with worse latency than the + original path. If a limited on-path attacker drops packets, the + original copy will still arrive at the destination endpoint. + + QUIC aims to constrain the capabilities of a limited off-path + attacker as follows: + + 1. A limited on-path attacker cannot cause a connection to close + once the handshake has completed. + + 2. A limited on-path attacker cannot cause an idle connection to + close if the client is first to resume activity. + + 3. A limited on-path attacker can cause an idle connection to be + deemed lost if the server is the first to resume activity. + + Note that these guarantees are the same guarantees provided for any + NAT, for the same reasons. + +21.2. Handshake Denial of Service + + As an encrypted and authenticated transport, QUIC provides a range of + protections against denial of service. Once the cryptographic + handshake is complete, QUIC endpoints discard most packets that are + not authenticated, greatly limiting the ability of an attacker to + interfere with existing connections. + + Once a connection is established, QUIC endpoints might accept some + unauthenticated ICMP packets (see Section 14.2.1), but the use of + these packets is extremely limited. The only other type of packet + that an endpoint might accept is a stateless reset (Section 10.3), + which relies on the token being kept secret until it is used. + + During the creation of a connection, QUIC only provides protection + against attacks from off the network path. All QUIC packets contain + proof that the recipient saw a preceding packet from its peer. + + Addresses cannot change during the handshake, so endpoints can + discard packets that are received on a different network path. + + The Source and Destination Connection ID fields are the primary means + of protection against an off-path attack during the handshake; see + Section 8.1. These are required to match those set by a peer. + Except for Initial and Stateless Resets, an endpoint only accepts + packets that include a Destination Connection ID field that matches a + value the endpoint previously chose. This is the only protection + offered for Version Negotiation packets. + + The Destination Connection ID field in an Initial packet is selected + by a client to be unpredictable, which serves an additional purpose. + The packets that carry the cryptographic handshake are protected with + a key that is derived from this connection ID and a salt specific to + the QUIC version. This allows endpoints to use the same process for + authenticating packets that they receive as they use after the + cryptographic handshake completes. Packets that cannot be + authenticated are discarded. Protecting packets in this fashion + provides a strong assurance that the sender of the packet saw the + Initial packet and understood it. + + These protections are not intended to be effective against an + attacker that is able to receive QUIC packets prior to the connection + being established. Such an attacker can potentially send packets + that will be accepted by QUIC endpoints. This version of QUIC + attempts to detect this sort of attack, but it expects that endpoints + will fail to establish a connection rather than recovering. For the + most part, the cryptographic handshake protocol [QUIC-TLS] is + responsible for detecting tampering during the handshake. + + Endpoints are permitted to use other methods to detect and attempt to + recover from interference with the handshake. Invalid packets can be + identified and discarded using other methods, but no specific method + is mandated in this document. + +21.3. Amplification Attack + + An attacker might be able to receive an address validation token + (Section 8) from a server and then release the IP address it used to + acquire that token. At a later time, the attacker can initiate a + 0-RTT connection with a server by spoofing this same address, which + might now address a different (victim) endpoint. The attacker can + thus potentially cause the server to send an initial congestion + window's worth of data towards the victim. + + Servers SHOULD provide mitigations for this attack by limiting the + usage and lifetime of address validation tokens; see Section 8.1.3. + +21.4. Optimistic ACK Attack + + An endpoint that acknowledges packets it has not received might cause + a congestion controller to permit sending at rates beyond what the + network supports. An endpoint MAY skip packet numbers when sending + packets to detect this behavior. An endpoint can then immediately + close the connection with a connection error of type + PROTOCOL_VIOLATION; see Section 10.2. + +21.5. Request Forgery Attacks + + A request forgery attack occurs where an endpoint causes its peer to + issue a request towards a victim, with the request controlled by the + endpoint. Request forgery attacks aim to provide an attacker with + access to capabilities of its peer that might otherwise be + unavailable to the attacker. For a networking protocol, a request + forgery attack is often used to exploit any implicit authorization + conferred on the peer by the victim due to the peer's location in the + network. + + For request forgery to be effective, an attacker needs to be able to + influence what packets the peer sends and where these packets are + sent. If an attacker can target a vulnerable service with a + controlled payload, that service might perform actions that are + attributed to the attacker's peer but are decided by the attacker. + + For example, cross-site request forgery [CSRF] exploits on the Web + cause a client to issue requests that include authorization cookies + [COOKIE], allowing one site access to information and actions that + are intended to be restricted to a different site. + + As QUIC runs over UDP, the primary attack modality of concern is one + where an attacker can select the address to which its peer sends UDP + datagrams and can control some of the unprotected content of those + packets. As much of the data sent by QUIC endpoints is protected, + this includes control over ciphertext. An attack is successful if an + attacker can cause a peer to send a UDP datagram to a host that will + perform some action based on content in the datagram. + + This section discusses ways in which QUIC might be used for request + forgery attacks. + + This section also describes limited countermeasures that can be + implemented by QUIC endpoints. These mitigations can be employed + unilaterally by a QUIC implementation or deployment, without + potential targets for request forgery attacks taking action. + However, these countermeasures could be insufficient if UDP-based + services do not properly authorize requests. + + Because the migration attack described in Section 21.5.4 is quite + powerful and does not have adequate countermeasures, QUIC server + implementations should assume that attackers can cause them to + generate arbitrary UDP payloads to arbitrary destinations. QUIC + servers SHOULD NOT be deployed in networks that do not deploy ingress + filtering [BCP38] and also have inadequately secured UDP endpoints. + + Although it is not generally possible to ensure that clients are not + co-located with vulnerable endpoints, this version of QUIC does not + allow servers to migrate, thus preventing spoofed migration attacks + on clients. Any future extension that allows server migration MUST + also define countermeasures for forgery attacks. + +21.5.1. Control Options for Endpoints + + QUIC offers some opportunities for an attacker to influence or + control where its peer sends UDP datagrams: + + * initial connection establishment (Section 7), where a server is + able to choose where a client sends datagrams -- for example, by + populating DNS records; + + * preferred addresses (Section 9.6), where a server is able to + choose where a client sends datagrams; + + * spoofed connection migrations (Section 9.3.1), where a client is + able to use source address spoofing to select where a server sends + subsequent datagrams; and + + * spoofed packets that cause a server to send a Version Negotiation + packet (Section 21.5.5). + + In all cases, the attacker can cause its peer to send datagrams to a + victim that might not understand QUIC. That is, these packets are + sent by the peer prior to address validation; see Section 8. + + Outside of the encrypted portion of packets, QUIC offers an endpoint + several options for controlling the content of UDP datagrams that its + peer sends. The Destination Connection ID field offers direct + control over bytes that appear early in packets sent by the peer; see + Section 5.1. The Token field in Initial packets offers a server + control over other bytes of Initial packets; see Section 17.2.2. + + There are no measures in this version of QUIC to prevent indirect + control over the encrypted portions of packets. It is necessary to + assume that endpoints are able to control the contents of frames that + a peer sends, especially those frames that convey application data, + such as STREAM frames. Though this depends to some degree on details + of the application protocol, some control is possible in many + protocol usage contexts. As the attacker has access to packet + protection keys, they are likely to be capable of predicting how a + peer will encrypt future packets. Successful control over datagram + content then only requires that the attacker be able to predict the + packet number and placement of frames in packets with some amount of + reliability. + + This section assumes that limiting control over datagram content is + not feasible. The focus of the mitigations in subsequent sections is + on limiting the ways in which datagrams that are sent prior to + address validation can be used for request forgery. + +21.5.2. Request Forgery with Client Initial Packets + + An attacker acting as a server can choose the IP address and port on + which it advertises its availability, so Initial packets from clients + are assumed to be available for use in this sort of attack. The + address validation implicit in the handshake ensures that -- for a + new connection -- a client will not send other types of packets to a + destination that does not understand QUIC or is not willing to accept + a QUIC connection. + + Initial packet protection (Section 5.2 of [QUIC-TLS]) makes it + difficult for servers to control the content of Initial packets sent + by clients. A client choosing an unpredictable Destination + Connection ID ensures that servers are unable to control any of the + encrypted portion of Initial packets from clients. + + However, the Token field is open to server control and does allow a + server to use clients to mount request forgery attacks. The use of + tokens provided with the NEW_TOKEN frame (Section 8.1.3) offers the + only option for request forgery during connection establishment. + + Clients, however, are not obligated to use the NEW_TOKEN frame. + Request forgery attacks that rely on the Token field can be avoided + if clients send an empty Token field when the server address has + changed from when the NEW_TOKEN frame was received. + + Clients could avoid using NEW_TOKEN if the server address changes. + However, not including a Token field could adversely affect + performance. Servers could rely on NEW_TOKEN to enable the sending + of data in excess of the three-times limit on sending data; see + Section 8.1. In particular, this affects cases where clients use + 0-RTT to request data from servers. + + Sending a Retry packet (Section 17.2.5) offers a server the option to + change the Token field. After sending a Retry, the server can also + control the Destination Connection ID field of subsequent Initial + packets from the client. This also might allow indirect control over + the encrypted content of Initial packets. However, the exchange of a + Retry packet validates the server's address, thereby preventing the + use of subsequent Initial packets for request forgery. + +21.5.3. Request Forgery with Preferred Addresses + + Servers can specify a preferred address, which clients then migrate + to after confirming the handshake; see Section 9.6. The Destination + Connection ID field of packets that the client sends to a preferred + address can be used for request forgery. + + A client MUST NOT send non-probing frames to a preferred address + prior to validating that address; see Section 8. This greatly + reduces the options that a server has to control the encrypted + portion of datagrams. + + This document does not offer any additional countermeasures that are + specific to the use of preferred addresses and can be implemented by + endpoints. The generic measures described in Section 21.5.6 could be + used as further mitigation. + +21.5.4. Request Forgery with Spoofed Migration + + Clients are able to present a spoofed source address as part of an + apparent connection migration to cause a server to send datagrams to + that address. + + The Destination Connection ID field in any packets that a server + subsequently sends to this spoofed address can be used for request + forgery. A client might also be able to influence the ciphertext. + + A server that only sends probing packets (Section 9.1) to an address + prior to address validation provides an attacker with only limited + control over the encrypted portion of datagrams. However, + particularly for NAT rebinding, this can adversely affect + performance. If the server sends frames carrying application data, + an attacker might be able to control most of the content of + datagrams. + + This document does not offer specific countermeasures that can be + implemented by endpoints, aside from the generic measures described + in Section 21.5.6. However, countermeasures for address spoofing at + the network level -- in particular, ingress filtering [BCP38] -- are + especially effective against attacks that use spoofing and originate + from an external network. + +21.5.5. Request Forgery with Version Negotiation + + Clients that are able to present a spoofed source address on a packet + can cause a server to send a Version Negotiation packet + (Section 17.2.1) to that address. + + The absence of size restrictions on the connection ID fields for + packets of an unknown version increases the amount of data that the + client controls from the resulting datagram. The first byte of this + packet is not under client control and the next four bytes are zero, + but the client is able to control up to 512 bytes starting from the + fifth byte. + + No specific countermeasures are provided for this attack, though + generic protections (Section 21.5.6) could apply. In this case, + ingress filtering [BCP38] is also effective. + +21.5.6. Generic Request Forgery Countermeasures + + The most effective defense against request forgery attacks is to + modify vulnerable services to use strong authentication. However, + this is not always something that is within the control of a QUIC + deployment. This section outlines some other steps that QUIC + endpoints could take unilaterally. These additional steps are all + discretionary because, depending on circumstances, they could + interfere with or prevent legitimate uses. + + Services offered over loopback interfaces often lack proper + authentication. Endpoints MAY prevent connection attempts or + migration to a loopback address. Endpoints SHOULD NOT allow + connections or migration to a loopback address if the same service + was previously available at a different interface or if the address + was provided by a service at a non-loopback address. Endpoints that + depend on these capabilities could offer an option to disable these + protections. + + Similarly, endpoints could regard a change in address to a link-local + address [RFC4291] or an address in a private-use range [RFC1918] from + a global, unique-local [RFC4193], or non-private address as a + potential attempt at request forgery. Endpoints could refuse to use + these addresses entirely, but that carries a significant risk of + interfering with legitimate uses. Endpoints SHOULD NOT refuse to use + an address unless they have specific knowledge about the network + indicating that sending datagrams to unvalidated addresses in a given + range is not safe. + + Endpoints MAY choose to reduce the risk of request forgery by not + including values from NEW_TOKEN frames in Initial packets or by only + sending probing frames in packets prior to completing address + validation. Note that this does not prevent an attacker from using + the Destination Connection ID field for an attack. + + Endpoints are not expected to have specific information about the + location of servers that could be vulnerable targets of a request + forgery attack. However, it might be possible over time to identify + specific UDP ports that are common targets of attacks or particular + patterns in datagrams that are used for attacks. Endpoints MAY + choose to avoid sending datagrams to these ports or not send + datagrams that match these patterns prior to validating the + destination address. Endpoints MAY retire connection IDs containing + patterns known to be problematic without using them. + + | Note: Modifying endpoints to apply these protections is more + | efficient than deploying network-based protections, as + | endpoints do not need to perform any additional processing when + | sending to an address that has been validated. + +21.6. Slowloris Attacks + + The attacks commonly known as Slowloris [SLOWLORIS] try to keep many + connections to the target endpoint open and hold them open as long as + possible. These attacks can be executed against a QUIC endpoint by + generating the minimum amount of activity necessary to avoid being + closed for inactivity. This might involve sending small amounts of + data, gradually opening flow control windows in order to control the + sender rate, or manufacturing ACK frames that simulate a high loss + rate. + + QUIC deployments SHOULD provide mitigations for the Slowloris + attacks, such as increasing the maximum number of clients the server + will allow, limiting the number of connections a single IP address is + allowed to make, imposing restrictions on the minimum transfer speed + a connection is allowed to have, and restricting the length of time + an endpoint is allowed to stay connected. + +21.7. Stream Fragmentation and Reassembly Attacks + + An adversarial sender might intentionally not send portions of the + stream data, causing the receiver to commit resources for the unsent + data. This could cause a disproportionate receive buffer memory + commitment and/or the creation of a large and inefficient data + structure at the receiver. + + An adversarial receiver might intentionally not acknowledge packets + containing stream data in an attempt to force the sender to store the + unacknowledged stream data for retransmission. + + The attack on receivers is mitigated if flow control windows + correspond to available memory. However, some receivers will + overcommit memory and advertise flow control offsets in the aggregate + that exceed actual available memory. The overcommitment strategy can + lead to better performance when endpoints are well behaved, but + renders endpoints vulnerable to the stream fragmentation attack. + + QUIC deployments SHOULD provide mitigations for stream fragmentation + attacks. Mitigations could consist of avoiding overcommitting + memory, limiting the size of tracking data structures, delaying + reassembly of STREAM frames, implementing heuristics based on the age + and duration of reassembly holes, or some combination of these. + +21.8. Stream Commitment Attack + + An adversarial endpoint can open a large number of streams, + exhausting state on an endpoint. The adversarial endpoint could + repeat the process on a large number of connections, in a manner + similar to SYN flooding attacks in TCP. + + Normally, clients will open streams sequentially, as explained in + Section 2.1. However, when several streams are initiated at short + intervals, loss or reordering can cause STREAM frames that open + streams to be received out of sequence. On receiving a higher- + numbered stream ID, a receiver is required to open all intervening + streams of the same type; see Section 3.2. Thus, on a new + connection, opening stream 4000000 opens 1 million and 1 client- + initiated bidirectional streams. + + The number of active streams is limited by the + initial_max_streams_bidi and initial_max_streams_uni transport + parameters as updated by any received MAX_STREAMS frames, as + explained in Section 4.6. If chosen judiciously, these limits + mitigate the effect of the stream commitment attack. However, + setting the limit too low could affect performance when applications + expect to open a large number of streams. + +21.9. Peer Denial of Service + + QUIC and TLS both contain frames or messages that have legitimate + uses in some contexts, but these frames or messages can be abused to + cause a peer to expend processing resources without having any + observable impact on the state of the connection. + + Messages can also be used to change and revert state in small or + inconsequential ways, such as by sending small increments to flow + control limits. + + If processing costs are disproportionately large in comparison to + bandwidth consumption or effect on state, then this could allow a + malicious peer to exhaust processing capacity. + + While there are legitimate uses for all messages, implementations + SHOULD track cost of processing relative to progress and treat + excessive quantities of any non-productive packets as indicative of + an attack. Endpoints MAY respond to this condition with a connection + error or by dropping packets. + +21.10. Explicit Congestion Notification Attacks + + An on-path attacker could manipulate the value of ECN fields in the + IP header to influence the sender's rate. [RFC3168] discusses + manipulations and their effects in more detail. + + A limited on-path attacker can duplicate and send packets with + modified ECN fields to affect the sender's rate. If duplicate + packets are discarded by a receiver, an attacker will need to race + the duplicate packet against the original to be successful in this + attack. Therefore, QUIC endpoints ignore the ECN field in an IP + packet unless at least one QUIC packet in that IP packet is + successfully processed; see Section 13.4. + +21.11. Stateless Reset Oracle + + Stateless resets create a possible denial-of-service attack analogous + to a TCP reset injection. This attack is possible if an attacker is + able to cause a stateless reset token to be generated for a + connection with a selected connection ID. An attacker that can cause + this token to be generated can reset an active connection with the + same connection ID. + + If a packet can be routed to different instances that share a static + key -- for example, by changing an IP address or port -- then an + attacker can cause the server to send a stateless reset. To defend + against this style of denial of service, endpoints that share a + static key for stateless resets (see Section 10.3.2) MUST be arranged + so that packets with a given connection ID always arrive at an + instance that has connection state, unless that connection is no + longer active. + + More generally, servers MUST NOT generate a stateless reset if a + connection with the corresponding connection ID could be active on + any endpoint using the same static key. + + In the case of a cluster that uses dynamic load balancing, it is + possible that a change in load-balancer configuration could occur + while an active instance retains connection state. Even if an + instance retains connection state, the change in routing and + resulting stateless reset will result in the connection being + terminated. If there is no chance of the packet being routed to the + correct instance, it is better to send a stateless reset than wait + for the connection to time out. However, this is acceptable only if + the routing cannot be influenced by an attacker. + +21.12. Version Downgrade + + This document defines QUIC Version Negotiation packets (Section 6), + which can be used to negotiate the QUIC version used between two + endpoints. However, this document does not specify how this + negotiation will be performed between this version and subsequent + future versions. In particular, Version Negotiation packets do not + contain any mechanism to prevent version downgrade attacks. Future + versions of QUIC that use Version Negotiation packets MUST define a + mechanism that is robust against version downgrade attacks. + +21.13. Targeted Attacks by Routing + + Deployments should limit the ability of an attacker to target a new + connection to a particular server instance. Ideally, routing + decisions are made independently of client-selected values, including + addresses. Once an instance is selected, a connection ID can be + selected so that later packets are routed to the same instance. + +21.14. Traffic Analysis + + The length of QUIC packets can reveal information about the length of + the content of those packets. The PADDING frame is provided so that + endpoints have some ability to obscure the length of packet content; + see Section 19.1. + + Defeating traffic analysis is challenging and the subject of active + research. Length is not the only way that information might leak. + Endpoints might also reveal sensitive information through other side + channels, such as the timing of packets. + +22. IANA Considerations + + This document establishes several registries for the management of + codepoints in QUIC. These registries operate on a common set of + policies as defined in Section 22.1. + +22.1. Registration Policies for QUIC Registries + + All QUIC registries allow for both provisional and permanent + registration of codepoints. This section documents policies that are + common to these registries. + +22.1.1. Provisional Registrations + + Provisional registrations of codepoints are intended to allow for + private use and experimentation with extensions to QUIC. Provisional + registrations only require the inclusion of the codepoint value and + contact information. However, provisional registrations could be + reclaimed and reassigned for another purpose. + + Provisional registrations require Expert Review, as defined in + Section 4.5 of [RFC8126]. The designated expert or experts are + advised that only registrations for an excessive proportion of + remaining codepoint space or the very first unassigned value (see + Section 22.1.2) can be rejected. + + Provisional registrations will include a Date field that indicates + when the registration was last updated. A request to update the date + on any provisional registration can be made without review from the + designated expert(s). + + All QUIC registries include the following fields to support + provisional registration: + + Value: The assigned codepoint. + Status: "permanent" or "provisional". + Specification: A reference to a publicly available specification for + the value. + Date: The date of the last update to the registration. + Change Controller: The entity that is responsible for the definition + of the registration. + Contact: Contact details for the registrant. + Notes: Supplementary notes about the registration. + + Provisional registrations MAY omit the Specification and Notes + fields, plus any additional fields that might be required for a + permanent registration. The Date field is not required as part of + requesting a registration, as it is set to the date the registration + is created or updated. + +22.1.2. Selecting Codepoints + + New requests for codepoints from QUIC registries SHOULD use a + randomly selected codepoint that excludes both existing allocations + and the first unallocated codepoint in the selected space. Requests + for multiple codepoints MAY use a contiguous range. This minimizes + the risk that differing semantics are attributed to the same + codepoint by different implementations. + + The use of the first unassigned codepoint is reserved for allocation + using the Standards Action policy; see Section 4.9 of [RFC8126]. The + early codepoint assignment process [EARLY-ASSIGN] can be used for + these values. + + For codepoints that are encoded in variable-length integers + (Section 16), such as frame types, codepoints that encode to four or + eight bytes (that is, values 2^14 and above) SHOULD be used unless + the usage is especially sensitive to having a longer encoding. + + Applications to register codepoints in QUIC registries MAY include a + requested codepoint as part of the registration. IANA MUST allocate + the selected codepoint if the codepoint is unassigned and the + requirements of the registration policy are met. + +22.1.3. Reclaiming Provisional Codepoints + + A request might be made to remove an unused provisional registration + from the registry to reclaim space in a registry, or a portion of the + registry (such as the 64-16383 range for codepoints that use + variable-length encodings). This SHOULD be done only for the + codepoints with the earliest recorded date, and entries that have + been updated less than a year prior SHOULD NOT be reclaimed. + + A request to remove a codepoint MUST be reviewed by the designated + experts. The experts MUST attempt to determine whether the codepoint + is still in use. Experts are advised to contact the listed contacts + for the registration, plus as wide a set of protocol implementers as + possible in order to determine whether any use of the codepoint is + known. The experts are also advised to allow at least four weeks for + responses. + + If any use of the codepoints is identified by this search or a + request to update the registration is made, the codepoint MUST NOT be + reclaimed. Instead, the date on the registration is updated. A note + might be added for the registration recording relevant information + that was learned. + + If no use of the codepoint was identified and no request was made to + update the registration, the codepoint MAY be removed from the + registry. + + This review and consultation process also applies to requests to + change a provisional registration into a permanent registration, + except that the goal is not to determine whether there is no use of + the codepoint but to determine that the registration is an accurate + representation of any deployed usage. + +22.1.4. Permanent Registrations + + Permanent registrations in QUIC registries use the Specification + Required policy (Section 4.6 of [RFC8126]), unless otherwise + specified. The designated expert or experts verify that a + specification exists and is readily accessible. Experts are + encouraged to be biased towards approving registrations unless they + are abusive, frivolous, or actively harmful (not merely aesthetically + displeasing or architecturally dubious). The creation of a registry + MAY specify additional constraints on permanent registrations. + + The creation of a registry MAY identify a range of codepoints where + registrations are governed by a different registration policy. For + instance, the "QUIC Frame Types" registry (Section 22.4) has a + stricter policy for codepoints in the range from 0 to 63. + + Any stricter requirements for permanent registrations do not prevent + provisional registrations for affected codepoints. For instance, a + provisional registration for a frame type of 61 could be requested. + + All registrations made by Standards Track publications MUST be + permanent. + + All registrations in this document are assigned a permanent status + and list a change controller of the IETF and a contact of the QUIC + Working Group (quic@ietf.org). + +22.2. QUIC Versions Registry + + IANA has added a registry for "QUIC Versions" under a "QUIC" heading. + + The "QUIC Versions" registry governs a 32-bit space; see Section 15. + This registry follows the registration policy from Section 22.1. + Permanent registrations in this registry are assigned using the + Specification Required policy (Section 4.6 of [RFC8126]). + + The codepoint of 0x00000001 for the protocol is assigned with + permanent status to the protocol defined in this document. The + codepoint of 0x00000000 is permanently reserved; the note for this + codepoint indicates that this version is reserved for version + negotiation. + + All codepoints that follow the pattern 0x?a?a?a?a are reserved, MUST + NOT be assigned by IANA, and MUST NOT appear in the listing of + assigned values. + +22.3. QUIC Transport Parameters Registry + + IANA has added a registry for "QUIC Transport Parameters" under a + "QUIC" heading. + + The "QUIC Transport Parameters" registry governs a 62-bit space. + This registry follows the registration policy from Section 22.1. + Permanent registrations in this registry are assigned using the + Specification Required policy (Section 4.6 of [RFC8126]), except for + values between 0x00 and 0x3f (in hexadecimal), inclusive, which are + assigned using Standards Action or IESG Approval as defined in + Sections 4.9 and 4.10 of [RFC8126]. + + In addition to the fields listed in Section 22.1.1, permanent + registrations in this registry MUST include the following field: + + Parameter Name: A short mnemonic for the parameter. + + The initial contents of this registry are shown in Table 6. + + +=======+=====================================+===============+ + | Value | Parameter Name | Specification | + +=======+=====================================+===============+ + | 0x00 | original_destination_connection_id | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x01 | max_idle_timeout | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x02 | stateless_reset_token | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x03 | max_udp_payload_size | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x04 | initial_max_data | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x05 | initial_max_stream_data_bidi_local | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x06 | initial_max_stream_data_bidi_remote | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x07 | initial_max_stream_data_uni | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x08 | initial_max_streams_bidi | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x09 | initial_max_streams_uni | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x0a | ack_delay_exponent | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x0b | max_ack_delay | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x0c | disable_active_migration | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x0d | preferred_address | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x0e | active_connection_id_limit | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x0f | initial_source_connection_id | Section 18.2 | + +-------+-------------------------------------+---------------+ + | 0x10 | retry_source_connection_id | Section 18.2 | + +-------+-------------------------------------+---------------+ + + Table 6: Initial QUIC Transport Parameters Registry Entries + + Each value of the form "31 * N + 27" for integer values of N (that + is, 27, 58, 89, ...) are reserved; these values MUST NOT be assigned + by IANA and MUST NOT appear in the listing of assigned values. + +22.4. QUIC Frame Types Registry + + IANA has added a registry for "QUIC Frame Types" under a "QUIC" + heading. + + The "QUIC Frame Types" registry governs a 62-bit space. This + registry follows the registration policy from Section 22.1. + Permanent registrations in this registry are assigned using the + Specification Required policy (Section 4.6 of [RFC8126]), except for + values between 0x00 and 0x3f (in hexadecimal), inclusive, which are + assigned using Standards Action or IESG Approval as defined in + Sections 4.9 and 4.10 of [RFC8126]. + + In addition to the fields listed in Section 22.1.1, permanent + registrations in this registry MUST include the following field: + + Frame Type Name: A short mnemonic for the frame type. + + In addition to the advice in Section 22.1, specifications for new + permanent registrations SHOULD describe the means by which an + endpoint might determine that it can send the identified type of + frame. An accompanying transport parameter registration is expected + for most registrations; see Section 22.3. Specifications for + permanent registrations also need to describe the format and assigned + semantics of any fields in the frame. + + The initial contents of this registry are tabulated in Table 3. Note + that the registry does not include the "Pkts" and "Spec" columns from + Table 3. + +22.5. QUIC Transport Error Codes Registry + + IANA has added a registry for "QUIC Transport Error Codes" under a + "QUIC" heading. + + The "QUIC Transport Error Codes" registry governs a 62-bit space. + This space is split into three ranges that are governed by different + policies. Permanent registrations in this registry are assigned + using the Specification Required policy (Section 4.6 of [RFC8126]), + except for values between 0x00 and 0x3f (in hexadecimal), inclusive, + which are assigned using Standards Action or IESG Approval as defined + in Sections 4.9 and 4.10 of [RFC8126]. + + In addition to the fields listed in Section 22.1.1, permanent + registrations in this registry MUST include the following fields: + + Code: A short mnemonic for the parameter. + + Description: A brief description of the error code semantics, which + MAY be a summary if a specification reference is provided. + + The initial contents of this registry are shown in Table 7. + + +=======+===========================+================+==============+ + |Value | Code |Description |Specification | + +=======+===========================+================+==============+ + |0x00 | NO_ERROR |No error |Section 20 | + +-------+---------------------------+----------------+--------------+ + |0x01 | INTERNAL_ERROR |Implementation |Section 20 | + | | |error | | + +-------+---------------------------+----------------+--------------+ + |0x02 | CONNECTION_REFUSED |Server refuses a|Section 20 | + | | |connection | | + +-------+---------------------------+----------------+--------------+ + |0x03 | FLOW_CONTROL_ERROR |Flow control |Section 20 | + | | |error | | + +-------+---------------------------+----------------+--------------+ + |0x04 | STREAM_LIMIT_ERROR |Too many streams|Section 20 | + | | |opened | | + +-------+---------------------------+----------------+--------------+ + |0x05 | STREAM_STATE_ERROR |Frame received |Section 20 | + | | |in invalid | | + | | |stream state | | + +-------+---------------------------+----------------+--------------+ + |0x06 | FINAL_SIZE_ERROR |Change to final |Section 20 | + | | |size | | + +-------+---------------------------+----------------+--------------+ + |0x07 | FRAME_ENCODING_ERROR |Frame encoding |Section 20 | + | | |error | | + +-------+---------------------------+----------------+--------------+ + |0x08 | TRANSPORT_PARAMETER_ERROR |Error in |Section 20 | + | | |transport | | + | | |parameters | | + +-------+---------------------------+----------------+--------------+ + |0x09 | CONNECTION_ID_LIMIT_ERROR |Too many |Section 20 | + | | |connection IDs | | + | | |received | | + +-------+---------------------------+----------------+--------------+ + |0x0a | PROTOCOL_VIOLATION |Generic protocol|Section 20 | + | | |violation | | + +-------+---------------------------+----------------+--------------+ + |0x0b | INVALID_TOKEN |Invalid Token |Section 20 | + | | |received | | + +-------+---------------------------+----------------+--------------+ + |0x0c | APPLICATION_ERROR |Application |Section 20 | + | | |error | | + +-------+---------------------------+----------------+--------------+ + |0x0d | CRYPTO_BUFFER_EXCEEDED |CRYPTO data |Section 20 | + | | |buffer | | + | | |overflowed | | + +-------+---------------------------+----------------+--------------+ + |0x0e | KEY_UPDATE_ERROR |Invalid packet |Section 20 | + | | |protection | | + | | |update | | + +-------+---------------------------+----------------+--------------+ + |0x0f | AEAD_LIMIT_REACHED |Excessive use of|Section 20 | + | | |packet | | + | | |protection keys | | + +-------+---------------------------+----------------+--------------+ + |0x10 | NO_VIABLE_PATH |No viable |Section 20 | + | | |network path | | + | | |exists | | + +-------+---------------------------+----------------+--------------+ + |0x0100-| CRYPTO_ERROR |TLS alert code |Section 20 | + |0x01ff | | | | + +-------+---------------------------+----------------+--------------+ + + Table 7: Initial QUIC Transport Error Codes Registry Entries + +23. References + +23.1. Normative References + + [BCP38] Ferguson, P. and D. Senie, "Network Ingress Filtering: + Defeating Denial of Service Attacks which employ IP Source + Address Spoofing", BCP 38, RFC 2827, May 2000. + + <https://www.rfc-editor.org/info/bcp38> + + [DPLPMTUD] Fairhurst, G., Jones, T., Tรผxen, M., Rรผngeler, I., and T. + Vรถlker, "Packetization Layer Path MTU Discovery for + Datagram Transports", RFC 8899, DOI 10.17487/RFC8899, + September 2020, <https://www.rfc-editor.org/info/rfc8899>. + + [EARLY-ASSIGN] + Cotton, M., "Early IANA Allocation of Standards Track Code + Points", BCP 100, RFC 7120, DOI 10.17487/RFC7120, January + 2014, <https://www.rfc-editor.org/info/rfc7120>. + + [IPv4] Postel, J., "Internet Protocol", STD 5, RFC 791, + DOI 10.17487/RFC0791, September 1981, + <https://www.rfc-editor.org/info/rfc791>. + + [QUIC-INVARIANTS] + Thomson, M., "Version-Independent Properties of QUIC", + RFC 8999, DOI 10.17487/RFC8999, May 2021, + <https://www.rfc-editor.org/info/rfc8999>. + + [QUIC-RECOVERY] + Iyengar, J., Ed. and I. Swett, Ed., "QUIC Loss Detection + and Congestion Control", RFC 9002, DOI 10.17487/RFC9002, + May 2021, <https://www.rfc-editor.org/info/rfc9002>. + + [QUIC-TLS] Thomson, M., Ed. and S. Turner, Ed., "Using TLS to Secure + QUIC", RFC 9001, DOI 10.17487/RFC9001, May 2021, + <https://www.rfc-editor.org/info/rfc9001>. + + [RFC1191] Mogul, J. and S. Deering, "Path MTU discovery", RFC 1191, + DOI 10.17487/RFC1191, November 1990, + <https://www.rfc-editor.org/info/rfc1191>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC3168] Ramakrishnan, K., Floyd, S., and D. Black, "The Addition + of Explicit Congestion Notification (ECN) to IP", + RFC 3168, DOI 10.17487/RFC3168, September 2001, + <https://www.rfc-editor.org/info/rfc3168>. + + [RFC3629] Yergeau, F., "UTF-8, a transformation format of ISO + 10646", STD 63, RFC 3629, DOI 10.17487/RFC3629, November + 2003, <https://www.rfc-editor.org/info/rfc3629>. + + [RFC6437] Amante, S., Carpenter, B., Jiang, S., and J. Rajahalme, + "IPv6 Flow Label Specification", RFC 6437, + DOI 10.17487/RFC6437, November 2011, + <https://www.rfc-editor.org/info/rfc6437>. + + [RFC8085] Eggert, L., Fairhurst, G., and G. Shepherd, "UDP Usage + Guidelines", BCP 145, RFC 8085, DOI 10.17487/RFC8085, + March 2017, <https://www.rfc-editor.org/info/rfc8085>. + + [RFC8126] Cotton, M., Leiba, B., and T. Narten, "Guidelines for + Writing an IANA Considerations Section in RFCs", BCP 26, + RFC 8126, DOI 10.17487/RFC8126, June 2017, + <https://www.rfc-editor.org/info/rfc8126>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + + [RFC8201] McCann, J., Deering, S., Mogul, J., and R. Hinden, Ed., + "Path MTU Discovery for IP version 6", STD 87, RFC 8201, + DOI 10.17487/RFC8201, July 2017, + <https://www.rfc-editor.org/info/rfc8201>. + + [RFC8311] Black, D., "Relaxing Restrictions on Explicit Congestion + Notification (ECN) Experimentation", RFC 8311, + DOI 10.17487/RFC8311, January 2018, + <https://www.rfc-editor.org/info/rfc8311>. + + [TLS13] Rescorla, E., "The Transport Layer Security (TLS) Protocol + Version 1.3", RFC 8446, DOI 10.17487/RFC8446, August 2018, + <https://www.rfc-editor.org/info/rfc8446>. + + [UDP] Postel, J., "User Datagram Protocol", STD 6, RFC 768, + DOI 10.17487/RFC0768, August 1980, + <https://www.rfc-editor.org/info/rfc768>. + +23.2. Informative References + + [AEAD] McGrew, D., "An Interface and Algorithms for Authenticated + Encryption", RFC 5116, DOI 10.17487/RFC5116, January 2008, + <https://www.rfc-editor.org/info/rfc5116>. + + [ALPN] Friedl, S., Popov, A., Langley, A., and E. Stephan, + "Transport Layer Security (TLS) Application-Layer Protocol + Negotiation Extension", RFC 7301, DOI 10.17487/RFC7301, + July 2014, <https://www.rfc-editor.org/info/rfc7301>. + + [ALTSVC] Nottingham, M., McManus, P., and J. Reschke, "HTTP + Alternative Services", RFC 7838, DOI 10.17487/RFC7838, + April 2016, <https://www.rfc-editor.org/info/rfc7838>. + + [COOKIE] Barth, A., "HTTP State Management Mechanism", RFC 6265, + DOI 10.17487/RFC6265, April 2011, + <https://www.rfc-editor.org/info/rfc6265>. + + [CSRF] Barth, A., Jackson, C., and J. Mitchell, "Robust defenses + for cross-site request forgery", Proceedings of the 15th + ACM conference on Computer and communications security - + CCS '08, DOI 10.1145/1455770.1455782, 2008, + <https://doi.org/10.1145/1455770.1455782>. + + [EARLY-DESIGN] + Roskind, J., "QUIC: Multiplexed Stream Transport Over + UDP", 2 December 2013, <https://docs.google.com/document/ + d/1RNHkx_VvKWyWg6Lr8SZ-saqsQx7rFV-ev2jRFUoVD34/ + edit?usp=sharing>. + + [GATEWAY] Hรคtรถnen, S., Nyrhinen, A., Eggert, L., Strowes, S., + Sarolahti, P., and M. Kojo, "An experimental study of home + gateway characteristics", Proceedings of the 10th ACM + SIGCOMM conference on Internet measurement - IMC '10, + DOI 10.1145/1879141.1879174, November 2010, + <https://doi.org/10.1145/1879141.1879174>. + + [HTTP2] Belshe, M., Peon, R., and M. Thomson, Ed., "Hypertext + Transfer Protocol Version 2 (HTTP/2)", RFC 7540, + DOI 10.17487/RFC7540, May 2015, + <https://www.rfc-editor.org/info/rfc7540>. + + [IPv6] Deering, S. and R. Hinden, "Internet Protocol, Version 6 + (IPv6) Specification", STD 86, RFC 8200, + DOI 10.17487/RFC8200, July 2017, + <https://www.rfc-editor.org/info/rfc8200>. + + [QUIC-MANAGEABILITY] + Kuehlewind, M. and B. Trammell, "Manageability of the QUIC + Transport Protocol", Work in Progress, Internet-Draft, + draft-ietf-quic-manageability-11, 21 April 2021, + <https://tools.ietf.org/html/draft-ietf-quic- + manageability-11>. + + [RANDOM] Eastlake 3rd, D., Schiller, J., and S. Crocker, + "Randomness Requirements for Security", BCP 106, RFC 4086, + DOI 10.17487/RFC4086, June 2005, + <https://www.rfc-editor.org/info/rfc4086>. + + [RFC1812] Baker, F., Ed., "Requirements for IP Version 4 Routers", + RFC 1812, DOI 10.17487/RFC1812, June 1995, + <https://www.rfc-editor.org/info/rfc1812>. + + [RFC1918] Rekhter, Y., Moskowitz, B., Karrenberg, D., de Groot, G. + J., and E. Lear, "Address Allocation for Private + Internets", BCP 5, RFC 1918, DOI 10.17487/RFC1918, + February 1996, <https://www.rfc-editor.org/info/rfc1918>. + + [RFC2018] Mathis, M., Mahdavi, J., Floyd, S., and A. Romanow, "TCP + Selective Acknowledgment Options", RFC 2018, + DOI 10.17487/RFC2018, October 1996, + <https://www.rfc-editor.org/info/rfc2018>. + + [RFC2104] Krawczyk, H., Bellare, M., and R. Canetti, "HMAC: Keyed- + Hashing for Message Authentication", RFC 2104, + DOI 10.17487/RFC2104, February 1997, + <https://www.rfc-editor.org/info/rfc2104>. + + [RFC3449] Balakrishnan, H., Padmanabhan, V., Fairhurst, G., and M. + Sooriyabandara, "TCP Performance Implications of Network + Path Asymmetry", BCP 69, RFC 3449, DOI 10.17487/RFC3449, + December 2002, <https://www.rfc-editor.org/info/rfc3449>. + + [RFC4193] Hinden, R. and B. Haberman, "Unique Local IPv6 Unicast + Addresses", RFC 4193, DOI 10.17487/RFC4193, October 2005, + <https://www.rfc-editor.org/info/rfc4193>. + + [RFC4291] Hinden, R. and S. Deering, "IP Version 6 Addressing + Architecture", RFC 4291, DOI 10.17487/RFC4291, February + 2006, <https://www.rfc-editor.org/info/rfc4291>. + + [RFC4443] Conta, A., Deering, S., and M. Gupta, Ed., "Internet + Control Message Protocol (ICMPv6) for the Internet + Protocol Version 6 (IPv6) Specification", STD 89, + RFC 4443, DOI 10.17487/RFC4443, March 2006, + <https://www.rfc-editor.org/info/rfc4443>. + + [RFC4787] Audet, F., Ed. and C. Jennings, "Network Address + Translation (NAT) Behavioral Requirements for Unicast + UDP", BCP 127, RFC 4787, DOI 10.17487/RFC4787, January + 2007, <https://www.rfc-editor.org/info/rfc4787>. + + [RFC5681] Allman, M., Paxson, V., and E. Blanton, "TCP Congestion + Control", RFC 5681, DOI 10.17487/RFC5681, September 2009, + <https://www.rfc-editor.org/info/rfc5681>. + + [RFC5869] Krawczyk, H. and P. Eronen, "HMAC-based Extract-and-Expand + Key Derivation Function (HKDF)", RFC 5869, + DOI 10.17487/RFC5869, May 2010, + <https://www.rfc-editor.org/info/rfc5869>. + + [RFC7983] Petit-Huguenin, M. and G. Salgueiro, "Multiplexing Scheme + Updates for Secure Real-time Transport Protocol (SRTP) + Extension for Datagram Transport Layer Security (DTLS)", + RFC 7983, DOI 10.17487/RFC7983, September 2016, + <https://www.rfc-editor.org/info/rfc7983>. + + [RFC8087] Fairhurst, G. and M. Welzl, "The Benefits of Using + Explicit Congestion Notification (ECN)", RFC 8087, + DOI 10.17487/RFC8087, March 2017, + <https://www.rfc-editor.org/info/rfc8087>. + + [RFC8981] Gont, F., Krishnan, S., Narten, T., and R. Draves, + "Temporary Address Extensions for Stateless Address + Autoconfiguration in IPv6", RFC 8981, + DOI 10.17487/RFC8981, February 2021, + <https://www.rfc-editor.org/info/rfc8981>. + + [SEC-CONS] Rescorla, E. and B. Korver, "Guidelines for Writing RFC + Text on Security Considerations", BCP 72, RFC 3552, + DOI 10.17487/RFC3552, July 2003, + <https://www.rfc-editor.org/info/rfc3552>. + + [SLOWLORIS] + "RSnake" Hansen, R., "Welcome to Slowloris - the low + bandwidth, yet greedy and poisonous HTTP client!", June + 2009, <https://web.archive.org/web/20150315054838/ + http://ha.ckers.org/slowloris/>. + +Appendix A. Pseudocode + + The pseudocode in this section describes sample algorithms. These + algorithms are intended to be correct and clear, rather than being + optimally performant. + + The pseudocode segments in this section are licensed as Code + Components; see the Copyright Notice. + +A.1. Sample Variable-Length Integer Decoding + + The pseudocode in Figure 45 shows how a variable-length integer can + be read from a stream of bytes. The function ReadVarint takes a + single argument -- a sequence of bytes, which can be read in network + byte order. + + ReadVarint(data): + // The length of variable-length integers is encoded in the + // first two bits of the first byte. + v = data.next_byte() + prefix = v >> 6 + length = 1 << prefix + + // Once the length is known, remove these bits and read any + // remaining bytes. + v = v & 0x3f + repeat length-1 times: + v = (v << 8) + data.next_byte() + return v + + Figure 45: Sample Variable-Length Integer Decoding Algorithm + + For example, the eight-byte sequence 0xc2197c5eff14e88c decodes to + the decimal value 151,288,809,941,952,652; the four-byte sequence + 0x9d7f3e7d decodes to 494,878,333; the two-byte sequence 0x7bbd + decodes to 15,293; and the single byte 0x25 decodes to 37 (as does + the two-byte sequence 0x4025). + +A.2. Sample Packet Number Encoding Algorithm + + The pseudocode in Figure 46 shows how an implementation can select an + appropriate size for packet number encodings. + + The EncodePacketNumber function takes two arguments: + + * full_pn is the full packet number of the packet being sent. + + * largest_acked is the largest packet number that has been + acknowledged by the peer in the current packet number space, if + any. + + EncodePacketNumber(full_pn, largest_acked): + + // The number of bits must be at least one more + // than the base-2 logarithm of the number of contiguous + // unacknowledged packet numbers, including the new packet. + if largest_acked is None: + num_unacked = full_pn + 1 + else: + num_unacked = full_pn - largest_acked + + min_bits = log(num_unacked, 2) + 1 + num_bytes = ceil(min_bits / 8) + + // Encode the integer value and truncate to + // the num_bytes least significant bytes. + return encode(full_pn, num_bytes) + + Figure 46: Sample Packet Number Encoding Algorithm + + For example, if an endpoint has received an acknowledgment for packet + 0xabe8b3 and is sending a packet with a number of 0xac5c02, there are + 29,519 (0x734f) outstanding packet numbers. In order to represent at + least twice this range (59,038 packets, or 0xe69e), 16 bits are + required. + + In the same state, sending a packet with a number of 0xace8fe uses + the 24-bit encoding, because at least 18 bits are required to + represent twice the range (131,222 packets, or 0x020096). + +A.3. Sample Packet Number Decoding Algorithm + + The pseudocode in Figure 47 includes an example algorithm for + decoding packet numbers after header protection has been removed. + + The DecodePacketNumber function takes three arguments: + + * largest_pn is the largest packet number that has been successfully + processed in the current packet number space. + + * truncated_pn is the value of the Packet Number field. + + * pn_nbits is the number of bits in the Packet Number field (8, 16, + 24, or 32). + + DecodePacketNumber(largest_pn, truncated_pn, pn_nbits): + expected_pn = largest_pn + 1 + pn_win = 1 << pn_nbits + pn_hwin = pn_win / 2 + pn_mask = pn_win - 1 + // The incoming packet number should be greater than + // expected_pn - pn_hwin and less than or equal to + // expected_pn + pn_hwin + // + // This means we cannot just strip the trailing bits from + // expected_pn and add the truncated_pn because that might + // yield a value outside the window. + // + // The following code calculates a candidate value and + // makes sure it's within the packet number window. + // Note the extra checks to prevent overflow and underflow. + candidate_pn = (expected_pn & ~pn_mask) | truncated_pn + if candidate_pn <= expected_pn - pn_hwin and + candidate_pn < (1 << 62) - pn_win: + return candidate_pn + pn_win + if candidate_pn > expected_pn + pn_hwin and + candidate_pn >= pn_win: + return candidate_pn - pn_win + return candidate_pn + + Figure 47: Sample Packet Number Decoding Algorithm + + For example, if the highest successfully authenticated packet had a + packet number of 0xa82f30ea, then a packet containing a 16-bit value + of 0x9b32 will be decoded as 0xa82f9b32. + +A.4. Sample ECN Validation Algorithm + + Each time an endpoint commences sending on a new network path, it + determines whether the path supports ECN; see Section 13.4. If the + path supports ECN, the goal is to use ECN. Endpoints might also + periodically reassess a path that was determined to not support ECN. + + This section describes one method for testing new paths. This + algorithm is intended to show how a path might be tested for ECN + support. Endpoints can implement different methods. + + The path is assigned an ECN state that is one of "testing", + "unknown", "failed", or "capable". On paths with a "testing" or + "capable" state, the endpoint sends packets with an ECT marking -- + ECT(0) by default; otherwise, the endpoint sends unmarked packets. + + To start testing a path, the ECN state is set to "testing", and + existing ECN counts are remembered as a baseline. + + The testing period runs for a number of packets or a limited time, as + determined by the endpoint. The goal is not to limit the duration of + the testing period but to ensure that enough marked packets are sent + for received ECN counts to provide a clear indication of how the path + treats marked packets. Section 13.4.2 suggests limiting this to ten + packets or three times the PTO. + + After the testing period ends, the ECN state for the path becomes + "unknown". From the "unknown" state, successful validation of the + ECN counts in an ACK frame (see Section 13.4.2.1) causes the ECN + state for the path to become "capable", unless no marked packet has + been acknowledged. + + If validation of ECN counts fails at any time, the ECN state for the + affected path becomes "failed". An endpoint can also mark the ECN + state for a path as "failed" if marked packets are all declared lost + or if they are all ECN-CE marked. + + Following this algorithm ensures that ECN is rarely disabled for + paths that properly support ECN. Any path that incorrectly modifies + markings will cause ECN to be disabled. For those rare cases where + marked packets are discarded by the path, the short duration of the + testing period limits the number of losses incurred. + +Contributors + + The original design and rationale behind this protocol draw + significantly from work by Jim Roskind [EARLY-DESIGN]. + + The IETF QUIC Working Group received an enormous amount of support + from many people. The following people provided substantive + contributions to this document: + + * Alessandro Ghedini + * Alyssa Wilk + * Antoine Delignat-Lavaud + * Brian Trammell + * Christian Huitema + * Colin Perkins + * David Schinazi + * Dmitri Tikhonov + * Eric Kinnear + * Eric Rescorla + * Gorry Fairhurst + * Ian Swett + * Igor Lubashev + * ๅฅฅ ไธ€็ฉ‚ (Kazuho Oku) + * Lars Eggert + * Lucas Pardue + * Magnus Westerlund + * Marten Seemann + * Martin Duke + * Mike Bishop + * Mikkel Fahnรธe Jรธrgensen + * Mirja Kรผhlewind + * Nick Banks + * Nick Harper + * Patrick McManus + * Roberto Peon + * Ryan Hamilton + * Subodh Iyengar + * Tatsuhiro Tsujikawa + * Ted Hardie + * Tom Jones + * Victor Vasiliev + +Authors' Addresses + + Jana Iyengar (editor) + Fastly + + Email: jri.ietf@gmail.com + + + Martin Thomson (editor) + Mozilla + + Email: mt@lowentropy.net diff --git a/eval/corpora/rfc/RFC 9001 - Using TLS to Secure QUIC.txt b/eval/corpora/rfc/RFC 9001 - Using TLS to Secure QUIC.txt new file mode 100644 index 00000000..331e4420 --- /dev/null +++ b/eval/corpora/rfc/RFC 9001 - Using TLS to Secure QUIC.txt @@ -0,0 +1,2756 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) M. Thomson, Ed. +Request for Comments: 9001 Mozilla +Category: Standards Track S. Turner, Ed. +ISSN: 2070-1721 sn3rd + May 2021 + + + Using TLS to Secure QUIC + +Abstract + + This document describes how Transport Layer Security (TLS) is used to + secure QUIC. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9001. + +Copyright Notice + + Copyright (c) 2021 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + +Table of Contents + + 1. Introduction + 2. Notational Conventions + 2.1. TLS Overview + 3. Protocol Overview + 4. Carrying TLS Messages + 4.1. Interface to TLS + 4.1.1. Handshake Complete + 4.1.2. Handshake Confirmed + 4.1.3. Sending and Receiving Handshake Messages + 4.1.4. Encryption Level Changes + 4.1.5. TLS Interface Summary + 4.2. TLS Version + 4.3. ClientHello Size + 4.4. Peer Authentication + 4.5. Session Resumption + 4.6. 0-RTT + 4.6.1. Enabling 0-RTT + 4.6.2. Accepting and Rejecting 0-RTT + 4.6.3. Validating 0-RTT Configuration + 4.7. HelloRetryRequest + 4.8. TLS Errors + 4.9. Discarding Unused Keys + 4.9.1. Discarding Initial Keys + 4.9.2. Discarding Handshake Keys + 4.9.3. Discarding 0-RTT Keys + 5. Packet Protection + 5.1. Packet Protection Keys + 5.2. Initial Secrets + 5.3. AEAD Usage + 5.4. Header Protection + 5.4.1. Header Protection Application + 5.4.2. Header Protection Sample + 5.4.3. AES-Based Header Protection + 5.4.4. ChaCha20-Based Header Protection + 5.5. Receiving Protected Packets + 5.6. Use of 0-RTT Keys + 5.7. Receiving Out-of-Order Protected Packets + 5.8. Retry Packet Integrity + 6. Key Update + 6.1. Initiating a Key Update + 6.2. Responding to a Key Update + 6.3. Timing of Receive Key Generation + 6.4. Sending with Updated Keys + 6.5. Receiving with Different Keys + 6.6. Limits on AEAD Usage + 6.7. Key Update Error Code + 7. Security of Initial Messages + 8. QUIC-Specific Adjustments to the TLS Handshake + 8.1. Protocol Negotiation + 8.2. QUIC Transport Parameters Extension + 8.3. Removing the EndOfEarlyData Message + 8.4. Prohibit TLS Middlebox Compatibility Mode + 9. Security Considerations + 9.1. Session Linkability + 9.2. Replay Attacks with 0-RTT + 9.3. Packet Reflection Attack Mitigation + 9.4. Header Protection Analysis + 9.5. Header Protection Timing Side Channels + 9.6. Key Diversity + 9.7. Randomness + 10. IANA Considerations + 11. References + 11.1. Normative References + 11.2. Informative References + Appendix A. Sample Packet Protection + A.1. Keys + A.2. Client Initial + A.3. Server Initial + A.4. Retry + A.5. ChaCha20-Poly1305 Short Header Packet + Appendix B. AEAD Algorithm Analysis + B.1. Analysis of AEAD_AES_128_GCM and AEAD_AES_256_GCM Usage + Limits + B.1.1. Confidentiality Limit + B.1.2. Integrity Limit + B.2. Analysis of AEAD_AES_128_CCM Usage Limits + Contributors + Authors' Addresses + +1. Introduction + + This document describes how QUIC [QUIC-TRANSPORT] is secured using + TLS [TLS13]. + + TLS 1.3 provides critical latency improvements for connection + establishment over previous versions. Absent packet loss, most new + connections can be established and secured within a single round + trip; on subsequent connections between the same client and server, + the client can often send application data immediately, that is, + using a zero round-trip setup. + + This document describes how TLS acts as a security component of QUIC. + +2. Notational Conventions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in BCP + 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + This document uses the terminology established in [QUIC-TRANSPORT]. + + For brevity, the acronym TLS is used to refer to TLS 1.3, though a + newer version could be used; see Section 4.2. + +2.1. TLS Overview + + TLS provides two endpoints with a way to establish a means of + communication over an untrusted medium (for example, the Internet). + TLS enables authentication of peers and provides confidentiality and + integrity protection for messages that endpoints exchange. + + Internally, TLS is a layered protocol, with the structure shown in + Figure 1. + + +-------------+------------+--------------+---------+ + Content | | | Application | | + Layer | Handshake | Alerts | Data | ... | + | | | | | + +-------------+------------+--------------+---------+ + Record | | + Layer | Records | + | | + +---------------------------------------------------+ + + Figure 1: TLS Layers + + Each content-layer message (e.g., handshake, alerts, and application + data) is carried as a series of typed TLS records by the record + layer. Records are individually cryptographically protected and then + transmitted over a reliable transport (typically TCP), which provides + sequencing and guaranteed delivery. + + The TLS authenticated key exchange occurs between two endpoints: + client and server. The client initiates the exchange and the server + responds. If the key exchange completes successfully, both client + and server will agree on a secret. TLS supports both pre-shared key + (PSK) and Diffie-Hellman over either finite fields or elliptic curves + ((EC)DHE) key exchanges. PSK is the basis for Early Data (0-RTT); + the latter provides forward secrecy (FS) when the (EC)DHE keys are + destroyed. The two modes can also be combined to provide forward + secrecy while using the PSK for authentication. + + After completing the TLS handshake, the client will have learned and + authenticated an identity for the server, and the server is + optionally able to learn and authenticate an identity for the client. + TLS supports X.509 [RFC5280] certificate-based authentication for + both server and client. When PSK key exchange is used (as in + resumption), knowledge of the PSK serves to authenticate the peer. + + The TLS key exchange is resistant to tampering by attackers, and it + produces shared secrets that cannot be controlled by either + participating peer. + + TLS provides two basic handshake modes of interest to QUIC: + + * A full 1-RTT handshake, in which the client is able to send + application data after one round trip and the server immediately + responds after receiving the first handshake message from the + client. + + * A 0-RTT handshake, in which the client uses information it has + previously learned about the server to send application data + immediately. This application data can be replayed by an + attacker, so 0-RTT is not suitable for carrying instructions that + might initiate any action that could cause unwanted effects if + replayed. + + A simplified TLS handshake with 0-RTT application data is shown in + Figure 2. + + Client Server + + ClientHello + (0-RTT Application Data) --------> + ServerHello + {EncryptedExtensions} + {Finished} + <-------- [Application Data] + {Finished} --------> + + [Application Data] <-------> [Application Data] + + () Indicates messages protected by Early Data (0-RTT) Keys + {} Indicates messages protected using Handshake Keys + [] Indicates messages protected using Application Data + (1-RTT) Keys + + Figure 2: TLS Handshake with 0-RTT + + Figure 2 omits the EndOfEarlyData message, which is not used in QUIC; + see Section 8.3. Likewise, neither ChangeCipherSpec nor KeyUpdate + messages are used by QUIC. ChangeCipherSpec is redundant in TLS 1.3; + see Section 8.4. QUIC has its own key update mechanism; see + Section 6. + + Data is protected using a number of encryption levels: + + * Initial keys + + * Early data (0-RTT) keys + + * Handshake keys + + * Application data (1-RTT) keys + + Application data can only appear in the early data and application + data levels. Handshake and alert messages may appear in any level. + + The 0-RTT handshake can be used if the client and server have + previously communicated. In the 1-RTT handshake, the client is + unable to send protected application data until it has received all + of the handshake messages sent by the server. + +3. Protocol Overview + + QUIC [QUIC-TRANSPORT] assumes responsibility for the confidentiality + and integrity protection of packets. For this it uses keys derived + from a TLS handshake [TLS13], but instead of carrying TLS records + over QUIC (as with TCP), TLS handshake and alert messages are carried + directly over the QUIC transport, which takes over the + responsibilities of the TLS record layer, as shown in Figure 3. + + +--------------+--------------+ +-------------+ + | TLS | TLS | | QUIC | + | Handshake | Alerts | | Applications| + | | | | (h3, etc.) | + +--------------+--------------+-+-------------+ + | | + | QUIC Transport | + | (streams, reliability, congestion, etc.) | + | | + +---------------------------------------------+ + | | + | QUIC Packet Protection | + | | + +---------------------------------------------+ + + Figure 3: QUIC Layers + + QUIC also relies on TLS for authentication and negotiation of + parameters that are critical to security and performance. + + Rather than a strict layering, these two protocols cooperate: QUIC + uses the TLS handshake; TLS uses the reliability, ordered delivery, + and record layer provided by QUIC. + + At a high level, there are two main interactions between the TLS and + QUIC components: + + * The TLS component sends and receives messages via the QUIC + component, with QUIC providing a reliable stream abstraction to + TLS. + + * The TLS component provides a series of updates to the QUIC + component, including (a) new packet protection keys to install and + (b) state changes such as handshake completion, the server + certificate, etc. + + Figure 4 shows these interactions in more detail, with the QUIC + packet protection being called out specially. + + +------------+ +------------+ + | |<---- Handshake Messages ----->| | + | |<- Validate 0-RTT Parameters ->| | + | |<--------- 0-RTT Keys ---------| | + | QUIC |<------- Handshake Keys -------| TLS | + | |<--------- 1-RTT Keys ---------| | + | |<------- Handshake Done -------| | + +------------+ +------------+ + | ^ + | Protect | Protected + v | Packet + +------------+ + | QUIC | + | Packet | + | Protection | + +------------+ + + Figure 4: QUIC and TLS Interactions + + Unlike TLS over TCP, QUIC applications that want to send data do not + send it using TLS Application Data records. Rather, they send it as + QUIC STREAM frames or other frame types, which are then carried in + QUIC packets. + +4. Carrying TLS Messages + + QUIC carries TLS handshake data in CRYPTO frames, each of which + consists of a contiguous block of handshake data identified by an + offset and length. Those frames are packaged into QUIC packets and + encrypted under the current encryption level. As with TLS over TCP, + once TLS handshake data has been delivered to QUIC, it is QUIC's + responsibility to deliver it reliably. Each chunk of data that is + produced by TLS is associated with the set of keys that TLS is + currently using. If QUIC needs to retransmit that data, it MUST use + the same keys even if TLS has already updated to newer keys. + + Each encryption level corresponds to a packet number space. The + packet number space that is used determines the semantics of frames. + Some frames are prohibited in different packet number spaces; see + Section 12.5 of [QUIC-TRANSPORT]. + + Because packets could be reordered on the wire, QUIC uses the packet + type to indicate which keys were used to protect a given packet, as + shown in Table 1. When packets of different types need to be sent, + endpoints SHOULD use coalesced packets to send them in the same UDP + datagram. + + +=====================+=================+==================+ + | Packet Type | Encryption Keys | PN Space | + +=====================+=================+==================+ + | Initial | Initial secrets | Initial | + +=====================+-----------------+------------------+ + | 0-RTT Protected | 0-RTT | Application data | + +=====================+-----------------+------------------+ + | Handshake | Handshake | Handshake | + +=====================+-----------------+------------------+ + | Retry | Retry | N/A | + +=====================+-----------------+------------------+ + | Version Negotiation | N/A | N/A | + +=====================+-----------------+------------------+ + | Short Header | 1-RTT | Application data | + +=====================+-----------------+------------------+ + + Table 1: Encryption Keys by Packet Type + + Section 17 of [QUIC-TRANSPORT] shows how packets at the various + encryption levels fit into the handshake process. + +4.1. Interface to TLS + + As shown in Figure 4, the interface from QUIC to TLS consists of four + primary functions: + + * Sending and receiving handshake messages + + * Processing stored transport and application state from a resumed + session and determining if it is valid to generate or accept 0-RTT + data + + * Rekeying (both transmit and receive) + + * Updating handshake state + + Additional functions might be needed to configure TLS. In + particular, QUIC and TLS need to agree on which is responsible for + validation of peer credentials, such as certificate validation + [RFC5280]. + +4.1.1. Handshake Complete + + In this document, the TLS handshake is considered complete when the + TLS stack has reported that the handshake is complete. This happens + when the TLS stack has both sent a Finished message and verified the + peer's Finished message. Verifying the peer's Finished message + provides the endpoints with an assurance that previous handshake + messages have not been modified. Note that the handshake does not + complete at both endpoints simultaneously. Consequently, any + requirement that is based on the completion of the handshake depends + on the perspective of the endpoint in question. + +4.1.2. Handshake Confirmed + + In this document, the TLS handshake is considered confirmed at the + server when the handshake completes. The server MUST send a + HANDSHAKE_DONE frame as soon as the handshake is complete. At the + client, the handshake is considered confirmed when a HANDSHAKE_DONE + frame is received. + + Additionally, a client MAY consider the handshake to be confirmed + when it receives an acknowledgment for a 1-RTT packet. This can be + implemented by recording the lowest packet number sent with 1-RTT + keys and comparing it to the Largest Acknowledged field in any + received 1-RTT ACK frame: once the latter is greater than or equal to + the former, the handshake is confirmed. + +4.1.3. Sending and Receiving Handshake Messages + + In order to drive the handshake, TLS depends on being able to send + and receive handshake messages. There are two basic functions on + this interface: one where QUIC requests handshake messages and one + where QUIC provides bytes that comprise handshake messages. + + Before starting the handshake, QUIC provides TLS with the transport + parameters (see Section 8.2) that it wishes to carry. + + A QUIC client starts TLS by requesting TLS handshake bytes from TLS. + The client acquires handshake bytes before sending its first packet. + A QUIC server starts the process by providing TLS with the client's + handshake bytes. + + At any time, the TLS stack at an endpoint will have a current sending + encryption level and a receiving encryption level. TLS encryption + levels determine the QUIC packet type and keys that are used for + protecting data. + + Each encryption level is associated with a different sequence of + bytes, which is reliably transmitted to the peer in CRYPTO frames. + When TLS provides handshake bytes to be sent, they are appended to + the handshake bytes for the current encryption level. The encryption + level then determines the type of packet that the resulting CRYPTO + frame is carried in; see Table 1. + + Four encryption levels are used, producing keys for Initial, 0-RTT, + Handshake, and 1-RTT packets. CRYPTO frames are carried in just + three of these levels, omitting the 0-RTT level. These four levels + correspond to three packet number spaces: Initial and Handshake + encrypted packets use their own separate spaces; 0-RTT and 1-RTT + packets use the application data packet number space. + + QUIC takes the unprotected content of TLS handshake records as the + content of CRYPTO frames. TLS record protection is not used by QUIC. + QUIC assembles CRYPTO frames into QUIC packets, which are protected + using QUIC packet protection. + + QUIC CRYPTO frames only carry TLS handshake messages. TLS alerts are + turned into QUIC CONNECTION_CLOSE error codes; see Section 4.8. TLS + application data and other content types cannot be carried by QUIC at + any encryption level; it is an error if they are received from the + TLS stack. + + When an endpoint receives a QUIC packet containing a CRYPTO frame + from the network, it proceeds as follows: + + * If the packet uses the current TLS receiving encryption level, + sequence the data into the input flow as usual. As with STREAM + frames, the offset is used to find the proper location in the data + sequence. If the result of this process is that new data is + available, then it is delivered to TLS in order. + + * If the packet is from a previously installed encryption level, it + MUST NOT contain data that extends past the end of previously + received data in that flow. Implementations MUST treat any + violations of this requirement as a connection error of type + PROTOCOL_VIOLATION. + + * If the packet is from a new encryption level, it is saved for + later processing by TLS. Once TLS moves to receiving from this + encryption level, saved data can be provided to TLS. When TLS + provides keys for a higher encryption level, if there is data from + a previous encryption level that TLS has not consumed, this MUST + be treated as a connection error of type PROTOCOL_VIOLATION. + + Each time that TLS is provided with new data, new handshake bytes are + requested from TLS. TLS might not provide any bytes if the handshake + messages it has received are incomplete or it has no data to send. + + The content of CRYPTO frames might either be processed incrementally + by TLS or buffered until complete messages or flights are available. + TLS is responsible for buffering handshake bytes that have arrived in + order. QUIC is responsible for buffering handshake bytes that arrive + out of order or for encryption levels that are not yet ready. QUIC + does not provide any means of flow control for CRYPTO frames; see + Section 7.5 of [QUIC-TRANSPORT]. + + Once the TLS handshake is complete, this is indicated to QUIC along + with any final handshake bytes that TLS needs to send. At this + stage, the transport parameters that the peer advertised during the + handshake are authenticated; see Section 8.2. + + Once the handshake is complete, TLS becomes passive. TLS can still + receive data from its peer and respond in kind, but it will not need + to send more data unless specifically requested -- either by an + application or QUIC. One reason to send data is that the server + might wish to provide additional or updated session tickets to a + client. + + When the handshake is complete, QUIC only needs to provide TLS with + any data that arrives in CRYPTO streams. In the same manner that is + used during the handshake, new data is requested from TLS after + providing received data. + +4.1.4. Encryption Level Changes + + As keys at a given encryption level become available to TLS, TLS + indicates to QUIC that reading or writing keys at that encryption + level are available. + + The availability of new keys is always a result of providing inputs + to TLS. TLS only provides new keys after being initialized (by a + client) or when provided with new handshake data. + + However, a TLS implementation could perform some of its processing + asynchronously. In particular, the process of validating a + certificate can take some time. While waiting for TLS processing to + complete, an endpoint SHOULD buffer received packets if they might be + processed using keys that are not yet available. These packets can + be processed once keys are provided by TLS. An endpoint SHOULD + continue to respond to packets that can be processed during this + time. + + After processing inputs, TLS might produce handshake bytes, keys for + new encryption levels, or both. + + TLS provides QUIC with three items as a new encryption level becomes + available: + + * A secret + + * An Authenticated Encryption with Associated Data (AEAD) function + + * A Key Derivation Function (KDF) + + These values are based on the values that TLS negotiates and are used + by QUIC to generate packet and header protection keys; see Section 5 + and Section 5.4. + + If 0-RTT is possible, it is ready after the client sends a TLS + ClientHello message or the server receives that message. After + providing a QUIC client with the first handshake bytes, the TLS stack + might signal the change to 0-RTT keys. On the server, after + receiving handshake bytes that contain a ClientHello message, a TLS + server might signal that 0-RTT keys are available. + + Although TLS only uses one encryption level at a time, QUIC may use + more than one level. For instance, after sending its Finished + message (using a CRYPTO frame at the Handshake encryption level) an + endpoint can send STREAM data (in 1-RTT encryption). If the Finished + message is lost, the endpoint uses the Handshake encryption level to + retransmit the lost message. Reordering or loss of packets can mean + that QUIC will need to handle packets at multiple encryption levels. + During the handshake, this means potentially handling packets at + higher and lower encryption levels than the current encryption level + used by TLS. + + In particular, server implementations need to be able to read packets + at the Handshake encryption level at the same time as the 0-RTT + encryption level. A client could interleave ACK frames that are + protected with Handshake keys with 0-RTT data, and the server needs + to process those acknowledgments in order to detect lost Handshake + packets. + + QUIC also needs access to keys that might not ordinarily be available + to a TLS implementation. For instance, a client might need to + acknowledge Handshake packets before it is ready to send CRYPTO + frames at that encryption level. TLS therefore needs to provide keys + to QUIC before it might produce them for its own use. + +4.1.5. TLS Interface Summary + + Figure 5 summarizes the exchange between QUIC and TLS for both client + and server. Solid arrows indicate packets that carry handshake data; + dashed arrows show where application data can be sent. Each arrow is + tagged with the encryption level used for that transmission. + + Client Server + ====== ====== + + Get Handshake + Initial -------------> + Install tx 0-RTT keys + 0-RTT - - - - - - - -> + + Handshake Received + Get Handshake + <------------- Initial + Install rx 0-RTT keys + Install Handshake keys + Get Handshake + <----------- Handshake + Install tx 1-RTT keys + <- - - - - - - - 1-RTT + + Handshake Received (Initial) + Install Handshake keys + Handshake Received (Handshake) + Get Handshake + Handshake -----------> + Handshake Complete + Install 1-RTT keys + 1-RTT - - - - - - - -> + + Handshake Received + Handshake Complete + Handshake Confirmed + Install rx 1-RTT keys + <--------------- 1-RTT + (HANDSHAKE_DONE) + Handshake Confirmed + + Figure 5: Interaction Summary between QUIC and TLS + + Figure 5 shows the multiple packets that form a single "flight" of + messages being processed individually, to show what incoming messages + trigger different actions. This shows multiple "Get Handshake" + invocations to retrieve handshake messages at different encryption + levels. New handshake messages are requested after incoming packets + have been processed. + + Figure 5 shows one possible structure for a simple handshake + exchange. The exact process varies based on the structure of + endpoint implementations and the order in which packets arrive. + Implementations could use a different number of operations or execute + them in other orders. + +4.2. TLS Version + + This document describes how TLS 1.3 [TLS13] is used with QUIC. + + In practice, the TLS handshake will negotiate a version of TLS to + use. This could result in a version of TLS newer than 1.3 being + negotiated if both endpoints support that version. This is + acceptable provided that the features of TLS 1.3 that are used by + QUIC are supported by the newer version. + + Clients MUST NOT offer TLS versions older than 1.3. A badly + configured TLS implementation could negotiate TLS 1.2 or another + older version of TLS. An endpoint MUST terminate the connection if a + version of TLS older than 1.3 is negotiated. + +4.3. ClientHello Size + + The first Initial packet from a client contains the start or all of + its first cryptographic handshake message, which for TLS is the + ClientHello. Servers might need to parse the entire ClientHello + (e.g., to access extensions such as Server Name Identification (SNI) + or Application-Layer Protocol Negotiation (ALPN)) in order to decide + whether to accept the new incoming QUIC connection. If the + ClientHello spans multiple Initial packets, such servers would need + to buffer the first received fragments, which could consume excessive + resources if the client's address has not yet been validated. To + avoid this, servers MAY use the Retry feature (see Section 8.1 of + [QUIC-TRANSPORT]) to only buffer partial ClientHello messages from + clients with a validated address. + + QUIC packet and framing add at least 36 bytes of overhead to the + ClientHello message. That overhead increases if the client chooses a + Source Connection ID field longer than zero bytes. Overheads also do + not include the token or a Destination Connection ID longer than 8 + bytes, both of which might be required if a server sends a Retry + packet. + + A typical TLS ClientHello can easily fit into a 1200-byte packet. + However, in addition to the overheads added by QUIC, there are + several variables that could cause this limit to be exceeded. Large + session tickets, multiple or large key shares, and long lists of + supported ciphers, signature algorithms, versions, QUIC transport + parameters, and other negotiable parameters and extensions could + cause this message to grow. + + For servers, in addition to connection IDs and tokens, the size of + TLS session tickets can have an effect on a client's ability to + connect efficiently. Minimizing the size of these values increases + the probability that clients can use them and still fit their entire + ClientHello message in their first Initial packet. + + The TLS implementation does not need to ensure that the ClientHello + is large enough to meet QUIC's requirements for datagrams that carry + Initial packets; see Section 14.1 of [QUIC-TRANSPORT]. QUIC + implementations use PADDING frames or packet coalescing to ensure + that datagrams are large enough. + +4.4. Peer Authentication + + The requirements for authentication depend on the application + protocol that is in use. TLS provides server authentication and + permits the server to request client authentication. + + A client MUST authenticate the identity of the server. This + typically involves verification that the identity of the server is + included in a certificate and that the certificate is issued by a + trusted entity (see for example [RFC2818]). + + | Note: Where servers provide certificates for authentication, + | the size of the certificate chain can consume a large number of + | bytes. Controlling the size of certificate chains is critical + | to performance in QUIC as servers are limited to sending 3 + | bytes for every byte received prior to validating the client + | address; see Section 8.1 of [QUIC-TRANSPORT]. The size of a + | certificate chain can be managed by limiting the number of + | names or extensions; using keys with small public key + | representations, like ECDSA; or by using certificate + | compression [COMPRESS]. + + A server MAY request that the client authenticate during the + handshake. A server MAY refuse a connection if the client is unable + to authenticate when requested. The requirements for client + authentication vary based on application protocol and deployment. + + A server MUST NOT use post-handshake client authentication (as + defined in Section 4.6.2 of [TLS13]) because the multiplexing offered + by QUIC prevents clients from correlating the certificate request + with the application-level event that triggered it (see + [HTTP2-TLS13]). More specifically, servers MUST NOT send post- + handshake TLS CertificateRequest messages, and clients MUST treat + receipt of such messages as a connection error of type + PROTOCOL_VIOLATION. + +4.5. Session Resumption + + QUIC can use the session resumption feature of TLS 1.3. It does this + by carrying NewSessionTicket messages in CRYPTO frames after the + handshake is complete. Session resumption can be used to provide + 0-RTT and can also be used when 0-RTT is disabled. + + Endpoints that use session resumption might need to remember some + information about the current connection when creating a resumed + connection. TLS requires that some information be retained; see + Section 4.6.1 of [TLS13]. QUIC itself does not depend on any state + being retained when resuming a connection unless 0-RTT is also used; + see Section 7.4.1 of [QUIC-TRANSPORT] and Section 4.6.1. Application + protocols could depend on state that is retained between resumed + connections. + + Clients can store any state required for resumption along with the + session ticket. Servers can use the session ticket to help carry + state. + + Session resumption allows servers to link activity on the original + connection with the resumed connection, which might be a privacy + issue for clients. Clients can choose not to enable resumption to + avoid creating this correlation. Clients SHOULD NOT reuse tickets as + that allows entities other than the server to correlate connections; + see Appendix C.4 of [TLS13]. + +4.6. 0-RTT + + The 0-RTT feature in QUIC allows a client to send application data + before the handshake is complete. This is made possible by reusing + negotiated parameters from a previous connection. To enable this, + 0-RTT depends on the client remembering critical parameters and + providing the server with a TLS session ticket that allows the server + to recover the same information. + + This information includes parameters that determine TLS state, as + governed by [TLS13], QUIC transport parameters, the chosen + application protocol, and any information the application protocol + might need; see Section 4.6.3. This information determines how 0-RTT + packets and their contents are formed. + + To ensure that the same information is available to both endpoints, + all information used to establish 0-RTT comes from the same + connection. Endpoints cannot selectively disregard information that + might alter the sending or processing of 0-RTT. + + [TLS13] sets a limit of seven days on the time between the original + connection and any attempt to use 0-RTT. There are other constraints + on 0-RTT usage, notably those caused by the potential exposure to + replay attack; see Section 9.2. + +4.6.1. Enabling 0-RTT + + The TLS early_data extension in the NewSessionTicket message is + defined to convey (in the max_early_data_size parameter) the amount + of TLS 0-RTT data the server is willing to accept. QUIC does not use + TLS early data. QUIC uses 0-RTT packets to carry early data. + Accordingly, the max_early_data_size parameter is repurposed to hold + a sentinel value 0xffffffff to indicate that the server is willing to + accept QUIC 0-RTT data. To indicate that the server does not accept + 0-RTT data, the early_data extension is omitted from the + NewSessionTicket. The amount of data that the client can send in + QUIC 0-RTT is controlled by the initial_max_data transport parameter + supplied by the server. + + Servers MUST NOT send the early_data extension with a + max_early_data_size field set to any value other than 0xffffffff. A + client MUST treat receipt of a NewSessionTicket that contains an + early_data extension with any other value as a connection error of + type PROTOCOL_VIOLATION. + + A client that wishes to send 0-RTT packets uses the early_data + extension in the ClientHello message of a subsequent handshake; see + Section 4.2.10 of [TLS13]. It then sends application data in 0-RTT + packets. + + A client that attempts 0-RTT might also provide an address validation + token if the server has sent a NEW_TOKEN frame; see Section 8.1 of + [QUIC-TRANSPORT]. + +4.6.2. Accepting and Rejecting 0-RTT + + A server accepts 0-RTT by sending an early_data extension in the + EncryptedExtensions; see Section 4.2.10 of [TLS13]. The server then + processes and acknowledges the 0-RTT packets that it receives. + + A server rejects 0-RTT by sending the EncryptedExtensions without an + early_data extension. A server will always reject 0-RTT if it sends + a TLS HelloRetryRequest. When rejecting 0-RTT, a server MUST NOT + process any 0-RTT packets, even if it could. When 0-RTT was + rejected, a client SHOULD treat receipt of an acknowledgment for a + 0-RTT packet as a connection error of type PROTOCOL_VIOLATION, if it + is able to detect the condition. + + When 0-RTT is rejected, all connection characteristics that the + client assumed might be incorrect. This includes the choice of + application protocol, transport parameters, and any application + configuration. The client therefore MUST reset the state of all + streams, including application state bound to those streams. + + A client MAY reattempt 0-RTT if it receives a Retry or Version + Negotiation packet. These packets do not signify rejection of 0-RTT. + +4.6.3. Validating 0-RTT Configuration + + When a server receives a ClientHello with the early_data extension, + it has to decide whether to accept or reject 0-RTT data from the + client. Some of this decision is made by the TLS stack (e.g., + checking that the cipher suite being resumed was included in the + ClientHello; see Section 4.2.10 of [TLS13]). Even when the TLS stack + has no reason to reject 0-RTT data, the QUIC stack or the application + protocol using QUIC might reject 0-RTT data because the configuration + of the transport or application associated with the resumed session + is not compatible with the server's current configuration. + + QUIC requires additional transport state to be associated with a + 0-RTT session ticket. One common way to implement this is using + stateless session tickets and storing this state in the session + ticket. Application protocols that use QUIC might have similar + requirements regarding associating or storing state. This associated + state is used for deciding whether 0-RTT data must be rejected. For + example, HTTP/3 settings [QUIC-HTTP] determine how 0-RTT data from + the client is interpreted. Other applications using QUIC could have + different requirements for determining whether to accept or reject + 0-RTT data. + +4.7. HelloRetryRequest + + The HelloRetryRequest message (see Section 4.1.4 of [TLS13]) can be + used to request that a client provide new information, such as a key + share, or to validate some characteristic of the client. From the + perspective of QUIC, HelloRetryRequest is not differentiated from + other cryptographic handshake messages that are carried in Initial + packets. Although it is in principle possible to use this feature + for address verification, QUIC implementations SHOULD instead use the + Retry feature; see Section 8.1 of [QUIC-TRANSPORT]. + +4.8. TLS Errors + + If TLS experiences an error, it generates an appropriate alert as + defined in Section 6 of [TLS13]. + + A TLS alert is converted into a QUIC connection error. The + AlertDescription value is added to 0x0100 to produce a QUIC error + code from the range reserved for CRYPTO_ERROR; see Section 20.1 of + [QUIC-TRANSPORT]. The resulting value is sent in a QUIC + CONNECTION_CLOSE frame of type 0x1c. + + QUIC is only able to convey an alert level of "fatal". In TLS 1.3, + the only existing uses for the "warning" level are to signal + connection close; see Section 6.1 of [TLS13]. As QUIC provides + alternative mechanisms for connection termination and the TLS + connection is only closed if an error is encountered, a QUIC endpoint + MUST treat any alert from TLS as if it were at the "fatal" level. + + QUIC permits the use of a generic code in place of a specific error + code; see Section 11 of [QUIC-TRANSPORT]. For TLS alerts, this + includes replacing any alert with a generic alert, such as + handshake_failure (0x0128 in QUIC). Endpoints MAY use a generic + error code to avoid possibly exposing confidential information. + +4.9. Discarding Unused Keys + + After QUIC has completed a move to a new encryption level, packet + protection keys for previous encryption levels can be discarded. + This occurs several times during the handshake, as well as when keys + are updated; see Section 6. + + Packet protection keys are not discarded immediately when new keys + are available. If packets from a lower encryption level contain + CRYPTO frames, frames that retransmit that data MUST be sent at the + same encryption level. Similarly, an endpoint generates + acknowledgments for packets at the same encryption level as the + packet being acknowledged. Thus, it is possible that keys for a + lower encryption level are needed for a short time after keys for a + newer encryption level are available. + + An endpoint cannot discard keys for a given encryption level unless + it has received all the cryptographic handshake messages from its + peer at that encryption level and its peer has done the same. + Different methods for determining this are provided for Initial keys + (Section 4.9.1) and Handshake keys (Section 4.9.2). These methods do + not prevent packets from being received or sent at that encryption + level because a peer might not have received all the acknowledgments + necessary. + + Though an endpoint might retain older keys, new data MUST be sent at + the highest currently available encryption level. Only ACK frames + and retransmissions of data in CRYPTO frames are sent at a previous + encryption level. These packets MAY also include PADDING frames. + +4.9.1. Discarding Initial Keys + + Packets protected with Initial secrets (Section 5.2) are not + authenticated, meaning that an attacker could spoof packets with the + intent to disrupt a connection. To limit these attacks, Initial + packet protection keys are discarded more aggressively than other + keys. + + The successful use of Handshake packets indicates that no more + Initial packets need to be exchanged, as these keys can only be + produced after receiving all CRYPTO frames from Initial packets. + Thus, a client MUST discard Initial keys when it first sends a + Handshake packet and a server MUST discard Initial keys when it first + successfully processes a Handshake packet. Endpoints MUST NOT send + Initial packets after this point. + + This results in abandoning loss recovery state for the Initial + encryption level and ignoring any outstanding Initial packets. + +4.9.2. Discarding Handshake Keys + + An endpoint MUST discard its Handshake keys when the TLS handshake is + confirmed (Section 4.1.2). + +4.9.3. Discarding 0-RTT Keys + + 0-RTT and 1-RTT packets share the same packet number space, and + clients do not send 0-RTT packets after sending a 1-RTT packet + (Section 5.6). + + Therefore, a client SHOULD discard 0-RTT keys as soon as it installs + 1-RTT keys as they have no use after that moment. + + Additionally, a server MAY discard 0-RTT keys as soon as it receives + a 1-RTT packet. However, due to packet reordering, a 0-RTT packet + could arrive after a 1-RTT packet. Servers MAY temporarily retain + 0-RTT keys to allow decrypting reordered packets without requiring + their contents to be retransmitted with 1-RTT keys. After receiving + a 1-RTT packet, servers MUST discard 0-RTT keys within a short time; + the RECOMMENDED time period is three times the Probe Timeout (PTO, + see [QUIC-RECOVERY]). A server MAY discard 0-RTT keys earlier if it + determines that it has received all 0-RTT packets, which can be done + by keeping track of missing packet numbers. + +5. Packet Protection + + As with TLS over TCP, QUIC protects packets with keys derived from + the TLS handshake, using the AEAD algorithm [AEAD] negotiated by TLS. + + QUIC packets have varying protections depending on their type: + + * Version Negotiation packets have no cryptographic protection. + + * Retry packets use AEAD_AES_128_GCM to provide protection against + accidental modification and to limit the entities that can produce + a valid Retry; see Section 5.8. + + * Initial packets use AEAD_AES_128_GCM with keys derived from the + Destination Connection ID field of the first Initial packet sent + by the client; see Section 5.2. + + * All other packets have strong cryptographic protections for + confidentiality and integrity, using keys and algorithms + negotiated by TLS. + + This section describes how packet protection is applied to Handshake + packets, 0-RTT packets, and 1-RTT packets. The same packet + protection process is applied to Initial packets. However, as it is + trivial to determine the keys used for Initial packets, these packets + are not considered to have confidentiality or integrity protection. + Retry packets use a fixed key and so similarly lack confidentiality + and integrity protection. + +5.1. Packet Protection Keys + + QUIC derives packet protection keys in the same way that TLS derives + record protection keys. + + Each encryption level has separate secret values for protection of + packets sent in each direction. These traffic secrets are derived by + TLS (see Section 7.1 of [TLS13]) and are used by QUIC for all + encryption levels except the Initial encryption level. The secrets + for the Initial encryption level are computed based on the client's + initial Destination Connection ID, as described in Section 5.2. + + The keys used for packet protection are computed from the TLS secrets + using the KDF provided by TLS. In TLS 1.3, the HKDF-Expand-Label + function described in Section 7.1 of [TLS13] is used with the hash + function from the negotiated cipher suite. All uses of HKDF-Expand- + Label in QUIC use a zero-length Context. + + Note that labels, which are described using strings, are encoded as + bytes using ASCII [ASCII] without quotes or any trailing NUL byte. + + Other versions of TLS MUST provide a similar function in order to be + used with QUIC. + + The current encryption level secret and the label "quic key" are + input to the KDF to produce the AEAD key; the label "quic iv" is used + to derive the Initialization Vector (IV); see Section 5.3. The + header protection key uses the "quic hp" label; see Section 5.4. + Using these labels provides key separation between QUIC and TLS; see + Section 9.6. + + Both "quic key" and "quic hp" are used to produce keys, so the Length + provided to HKDF-Expand-Label along with these labels is determined + by the size of keys in the AEAD or header protection algorithm. The + Length provided with "quic iv" is the minimum length of the AEAD + nonce or 8 bytes if that is larger; see [AEAD]. + + The KDF used for initial secrets is always the HKDF-Expand-Label + function from TLS 1.3; see Section 5.2. + +5.2. Initial Secrets + + Initial packets apply the packet protection process, but use a secret + derived from the Destination Connection ID field from the client's + first Initial packet. + + This secret is determined by using HKDF-Extract (see Section 2.2 of + [HKDF]) with a salt of 0x38762cf7f55934b34d179ae6a4c80cadccbb7f0a and + the input keying material (IKM) of the Destination Connection ID + field. This produces an intermediate pseudorandom key (PRK) that is + used to derive two separate secrets for sending and receiving. + + The secret used by clients to construct Initial packets uses the PRK + and the label "client in" as input to the HKDF-Expand-Label function + from TLS [TLS13] to produce a 32-byte secret. Packets constructed by + the server use the same process with the label "server in". The hash + function for HKDF when deriving initial secrets and keys is SHA-256 + [SHA]. + + This process in pseudocode is: + + initial_salt = 0x38762cf7f55934b34d179ae6a4c80cadccbb7f0a + initial_secret = HKDF-Extract(initial_salt, + client_dst_connection_id) + + client_initial_secret = HKDF-Expand-Label(initial_secret, + "client in", "", + Hash.length) + server_initial_secret = HKDF-Expand-Label(initial_secret, + "server in", "", + Hash.length) + + The connection ID used with HKDF-Expand-Label is the Destination + Connection ID in the Initial packet sent by the client. This will be + a randomly selected value unless the client creates the Initial + packet after receiving a Retry packet, where the Destination + Connection ID is selected by the server. + + Future versions of QUIC SHOULD generate a new salt value, thus + ensuring that the keys are different for each version of QUIC. This + prevents a middlebox that recognizes only one version of QUIC from + seeing or modifying the contents of packets from future versions. + + The HKDF-Expand-Label function defined in TLS 1.3 MUST be used for + Initial packets even where the TLS versions offered do not include + TLS 1.3. + + The secrets used for constructing subsequent Initial packets change + when a server sends a Retry packet to use the connection ID value + selected by the server. The secrets do not change when a client + changes the Destination Connection ID it uses in response to an + Initial packet from the server. + + | Note: The Destination Connection ID field could be any length + | up to 20 bytes, including zero length if the server sends a + | Retry packet with a zero-length Source Connection ID field. + | After a Retry, the Initial keys provide the client no assurance + | that the server received its packet, so the client has to rely + | on the exchange that included the Retry packet to validate the + | server address; see Section 8.1 of [QUIC-TRANSPORT]. + + Appendix A contains sample Initial packets. + +5.3. AEAD Usage + + The Authenticated Encryption with Associated Data (AEAD) function + (see [AEAD]) used for QUIC packet protection is the AEAD that is + negotiated for use with the TLS connection. For example, if TLS is + using the TLS_AES_128_GCM_SHA256 cipher suite, the AEAD_AES_128_GCM + function is used. + + QUIC can use any of the cipher suites defined in [TLS13] with the + exception of TLS_AES_128_CCM_8_SHA256. A cipher suite MUST NOT be + negotiated unless a header protection scheme is defined for the + cipher suite. This document defines a header protection scheme for + all cipher suites defined in [TLS13] aside from + TLS_AES_128_CCM_8_SHA256. These cipher suites have a 16-byte + authentication tag and produce an output 16 bytes larger than their + input. + + An endpoint MUST NOT reject a ClientHello that offers a cipher suite + that it does not support, or it would be impossible to deploy a new + cipher suite. This also applies to TLS_AES_128_CCM_8_SHA256. + + When constructing packets, the AEAD function is applied prior to + applying header protection; see Section 5.4. The unprotected packet + header is part of the associated data (A). When processing packets, + an endpoint first removes the header protection. + + The key and IV for the packet are computed as described in + Section 5.1. The nonce, N, is formed by combining the packet + protection IV with the packet number. The 62 bits of the + reconstructed QUIC packet number in network byte order are left- + padded with zeros to the size of the IV. The exclusive OR of the + padded packet number and the IV forms the AEAD nonce. + + The associated data, A, for the AEAD is the contents of the QUIC + header, starting from the first byte of either the short or long + header, up to and including the unprotected packet number. + + The input plaintext, P, for the AEAD is the payload of the QUIC + packet, as described in [QUIC-TRANSPORT]. + + The output ciphertext, C, of the AEAD is transmitted in place of P. + + Some AEAD functions have limits for how many packets can be encrypted + under the same key and IV; see Section 6.6. This might be lower than + the packet number limit. An endpoint MUST initiate a key update + (Section 6) prior to exceeding any limit set for the AEAD that is in + use. + +5.4. Header Protection + + Parts of QUIC packet headers, in particular the Packet Number field, + are protected using a key that is derived separately from the packet + protection key and IV. The key derived using the "quic hp" label is + used to provide confidentiality protection for those fields that are + not exposed to on-path elements. + + This protection applies to the least significant bits of the first + byte, plus the Packet Number field. The four least significant bits + of the first byte are protected for packets with long headers; the + five least significant bits of the first byte are protected for + packets with short headers. For both header forms, this covers the + reserved bits and the Packet Number Length field; the Key Phase bit + is also protected for packets with a short header. + + The same header protection key is used for the duration of the + connection, with the value not changing after a key update (see + Section 6). This allows header protection to be used to protect the + key phase. + + This process does not apply to Retry or Version Negotiation packets, + which do not contain a protected payload or any of the fields that + are protected by this process. + +5.4.1. Header Protection Application + + Header protection is applied after packet protection is applied (see + Section 5.3). The ciphertext of the packet is sampled and used as + input to an encryption algorithm. The algorithm used depends on the + negotiated AEAD. + + The output of this algorithm is a 5-byte mask that is applied to the + protected header fields using exclusive OR. The least significant + bits of the first byte of the packet are masked by the least + significant bits of the first mask byte, and the packet number is + masked with the remaining bytes. Any unused bytes of mask that might + result from a shorter packet number encoding are unused. + + Figure 6 shows a sample algorithm for applying header protection. + Removing header protection only differs in the order in which the + packet number length (pn_length) is determined (here "^" is used to + represent exclusive OR). + + mask = header_protection(hp_key, sample) + + pn_length = (packet[0] & 0x03) + 1 + if (packet[0] & 0x80) == 0x80: + # Long header: 4 bits masked + packet[0] ^= mask[0] & 0x0f + else: + # Short header: 5 bits masked + packet[0] ^= mask[0] & 0x1f + + # pn_offset is the start of the Packet Number field. + packet[pn_offset:pn_offset+pn_length] ^= mask[1:1+pn_length] + + Figure 6: Header Protection Pseudocode + + Specific header protection functions are defined based on the + selected cipher suite; see Section 5.4.3 and Section 5.4.4. + + Figure 7 shows an example long header packet (Initial) and a short + header packet (1-RTT). Figure 7 shows the fields in each header that + are covered by header protection and the portion of the protected + packet payload that is sampled. + + Initial Packet { + Header Form (1) = 1, + Fixed Bit (1) = 1, + Long Packet Type (2) = 0, + Reserved Bits (2), # Protected + Packet Number Length (2), # Protected + Version (32), + DCID Len (8), + Destination Connection ID (0..160), + SCID Len (8), + Source Connection ID (0..160), + Token Length (i), + Token (..), + Length (i), + Packet Number (8..32), # Protected + Protected Payload (0..24), # Skipped Part + Protected Payload (128), # Sampled Part + Protected Payload (..) # Remainder + } + + 1-RTT Packet { + Header Form (1) = 0, + Fixed Bit (1) = 1, + Spin Bit (1), + Reserved Bits (2), # Protected + Key Phase (1), # Protected + Packet Number Length (2), # Protected + Destination Connection ID (0..160), + Packet Number (8..32), # Protected + Protected Payload (0..24), # Skipped Part + Protected Payload (128), # Sampled Part + Protected Payload (..), # Remainder + } + + Figure 7: Header Protection and Ciphertext Sample + + Before a TLS cipher suite can be used with QUIC, a header protection + algorithm MUST be specified for the AEAD used with that cipher suite. + This document defines algorithms for AEAD_AES_128_GCM, + AEAD_AES_128_CCM, AEAD_AES_256_GCM (all these AES AEADs are defined + in [AEAD]), and AEAD_CHACHA20_POLY1305 (defined in [CHACHA]). Prior + to TLS selecting a cipher suite, AES header protection is used + (Section 5.4.3), matching the AEAD_AES_128_GCM packet protection. + +5.4.2. Header Protection Sample + + The header protection algorithm uses both the header protection key + and a sample of the ciphertext from the packet Payload field. + + The same number of bytes are always sampled, but an allowance needs + to be made for the removal of protection by a receiving endpoint, + which will not know the length of the Packet Number field. The + sample of ciphertext is taken starting from an offset of 4 bytes + after the start of the Packet Number field. That is, in sampling + packet ciphertext for header protection, the Packet Number field is + assumed to be 4 bytes long (its maximum possible encoded length). + + An endpoint MUST discard packets that are not long enough to contain + a complete sample. + + To ensure that sufficient data is available for sampling, packets are + padded so that the combined lengths of the encoded packet number and + protected payload is at least 4 bytes longer than the sample required + for header protection. The cipher suites defined in [TLS13] -- other + than TLS_AES_128_CCM_8_SHA256, for which a header protection scheme + is not defined in this document -- have 16-byte expansions and + 16-byte header protection samples. This results in needing at least + 3 bytes of frames in the unprotected payload if the packet number is + encoded on a single byte, or 2 bytes of frames for a 2-byte packet + number encoding. + + The sampled ciphertext can be determined by the following pseudocode: + + # pn_offset is the start of the Packet Number field. + sample_offset = pn_offset + 4 + + sample = packet[sample_offset..sample_offset+sample_length] + + Where the packet number offset of a short header packet can be + calculated as: + + pn_offset = 1 + len(connection_id) + + And the packet number offset of a long header packet can be + calculated as: + + pn_offset = 7 + len(destination_connection_id) + + len(source_connection_id) + + len(payload_length) + if packet_type == Initial: + pn_offset += len(token_length) + + len(token) + + For example, for a packet with a short header, an 8-byte connection + ID, and protected with AEAD_AES_128_GCM, the sample takes bytes 13 to + 28 inclusive (using zero-based indexing). + + Multiple QUIC packets might be included in the same UDP datagram. + Each packet is handled separately. + +5.4.3. AES-Based Header Protection + + This section defines the packet protection algorithm for + AEAD_AES_128_GCM, AEAD_AES_128_CCM, and AEAD_AES_256_GCM. + AEAD_AES_128_GCM and AEAD_AES_128_CCM use 128-bit AES in Electronic + Codebook (ECB) mode. AEAD_AES_256_GCM uses 256-bit AES in ECB mode. + AES is defined in [AES]. + + This algorithm samples 16 bytes from the packet ciphertext. This + value is used as the input to AES-ECB. In pseudocode, the header + protection function is defined as: + + header_protection(hp_key, sample): + mask = AES-ECB(hp_key, sample) + +5.4.4. ChaCha20-Based Header Protection + + When AEAD_CHACHA20_POLY1305 is in use, header protection uses the raw + ChaCha20 function as defined in Section 2.4 of [CHACHA]. This uses a + 256-bit key and 16 bytes sampled from the packet protection output. + + The first 4 bytes of the sampled ciphertext are the block counter. A + ChaCha20 implementation could take a 32-bit integer in place of a + byte sequence, in which case, the byte sequence is interpreted as a + little-endian value. + + The remaining 12 bytes are used as the nonce. A ChaCha20 + implementation might take an array of three 32-bit integers in place + of a byte sequence, in which case, the nonce bytes are interpreted as + a sequence of 32-bit little-endian integers. + + The encryption mask is produced by invoking ChaCha20 to protect 5 + zero bytes. In pseudocode, the header protection function is defined + as: + + header_protection(hp_key, sample): + counter = sample[0..3] + nonce = sample[4..15] + mask = ChaCha20(hp_key, counter, nonce, {0,0,0,0,0}) + +5.5. Receiving Protected Packets + + Once an endpoint successfully receives a packet with a given packet + number, it MUST discard all packets in the same packet number space + with higher packet numbers if they cannot be successfully unprotected + with either the same key, or -- if there is a key update -- a + subsequent packet protection key; see Section 6. Similarly, a packet + that appears to trigger a key update but cannot be unprotected + successfully MUST be discarded. + + Failure to unprotect a packet does not necessarily indicate the + existence of a protocol error in a peer or an attack. The truncated + packet number encoding used in QUIC can cause packet numbers to be + decoded incorrectly if they are delayed significantly. + +5.6. Use of 0-RTT Keys + + If 0-RTT keys are available (see Section 4.6.1), the lack of replay + protection means that restrictions on their use are necessary to + avoid replay attacks on the protocol. + + Of the frames defined in [QUIC-TRANSPORT], the STREAM, RESET_STREAM, + STOP_SENDING, and CONNECTION_CLOSE frames are potentially unsafe for + use with 0-RTT as they carry application data. Application data that + is received in 0-RTT could cause an application at the server to + process the data multiple times rather than just once. Additional + actions taken by a server as a result of processing replayed + application data could have unwanted consequences. A client + therefore MUST NOT use 0-RTT for application data unless specifically + requested by the application that is in use. + + An application protocol that uses QUIC MUST include a profile that + defines acceptable use of 0-RTT; otherwise, 0-RTT can only be used to + carry QUIC frames that do not carry application data. For example, a + profile for HTTP is described in [HTTP-REPLAY] and used for HTTP/3; + see Section 10.9 of [QUIC-HTTP]. + + Though replaying packets might result in additional connection + attempts, the effect of processing replayed frames that do not carry + application data is limited to changing the state of the affected + connection. A TLS handshake cannot be successfully completed using + replayed packets. + + A client MAY wish to apply additional restrictions on what data it + sends prior to the completion of the TLS handshake. + + A client otherwise treats 0-RTT keys as equivalent to 1-RTT keys, + except that it cannot send certain frames with 0-RTT keys; see + Section 12.5 of [QUIC-TRANSPORT]. + + A client that receives an indication that its 0-RTT data has been + accepted by a server can send 0-RTT data until it receives all of the + server's handshake messages. A client SHOULD stop sending 0-RTT data + if it receives an indication that 0-RTT data has been rejected. + + A server MUST NOT use 0-RTT keys to protect packets; it uses 1-RTT + keys to protect acknowledgments of 0-RTT packets. A client MUST NOT + attempt to decrypt 0-RTT packets it receives and instead MUST discard + them. + + Once a client has installed 1-RTT keys, it MUST NOT send any more + 0-RTT packets. + + | Note: 0-RTT data can be acknowledged by the server as it + | receives it, but any packets containing acknowledgments of + | 0-RTT data cannot have packet protection removed by the client + | until the TLS handshake is complete. The 1-RTT keys necessary + | to remove packet protection cannot be derived until the client + | receives all server handshake messages. + +5.7. Receiving Out-of-Order Protected Packets + + Due to reordering and loss, protected packets might be received by an + endpoint before the final TLS handshake messages are received. A + client will be unable to decrypt 1-RTT packets from the server, + whereas a server will be able to decrypt 1-RTT packets from the + client. Endpoints in either role MUST NOT decrypt 1-RTT packets from + their peer prior to completing the handshake. + + Even though 1-RTT keys are available to a server after receiving the + first handshake messages from a client, it is missing assurances on + the client state: + + * The client is not authenticated, unless the server has chosen to + use a pre-shared key and validated the client's pre-shared key + binder; see Section 4.2.11 of [TLS13]. + + * The client has not demonstrated liveness, unless the server has + validated the client's address with a Retry packet or other means; + see Section 8.1 of [QUIC-TRANSPORT]. + + * Any received 0-RTT data that the server responds to might be due + to a replay attack. + + Therefore, the server's use of 1-RTT keys before the handshake is + complete is limited to sending data. A server MUST NOT process + incoming 1-RTT protected packets before the TLS handshake is + complete. Because sending acknowledgments indicates that all frames + in a packet have been processed, a server cannot send acknowledgments + for 1-RTT packets until the TLS handshake is complete. Received + packets protected with 1-RTT keys MAY be stored and later decrypted + and used once the handshake is complete. + + | Note: TLS implementations might provide all 1-RTT secrets prior + | to handshake completion. Even where QUIC implementations have + | 1-RTT read keys, those keys are not to be used prior to + | completing the handshake. + + The requirement for the server to wait for the client Finished + message creates a dependency on that message being delivered. A + client can avoid the potential for head-of-line blocking that this + implies by sending its 1-RTT packets coalesced with a Handshake + packet containing a copy of the CRYPTO frame that carries the + Finished message, until one of the Handshake packets is acknowledged. + This enables immediate server processing for those packets. + + A server could receive packets protected with 0-RTT keys prior to + receiving a TLS ClientHello. The server MAY retain these packets for + later decryption in anticipation of receiving a ClientHello. + + A client generally receives 1-RTT keys at the same time as the + handshake completes. Even if it has 1-RTT secrets, a client MUST NOT + process incoming 1-RTT protected packets before the TLS handshake is + complete. + +5.8. Retry Packet Integrity + + Retry packets (see Section 17.2.5 of [QUIC-TRANSPORT]) carry a Retry + Integrity Tag that provides two properties: it allows the discarding + of packets that have accidentally been corrupted by the network, and + only an entity that observes an Initial packet can send a valid Retry + packet. + + The Retry Integrity Tag is a 128-bit field that is computed as the + output of AEAD_AES_128_GCM [AEAD] used with the following inputs: + + * The secret key, K, is 128 bits equal to + 0xbe0c690b9f66575a1d766b54e368c84e. + + * The nonce, N, is 96 bits equal to 0x461599d35d632bf2239825bb. + + * The plaintext, P, is empty. + + * The associated data, A, is the contents of the Retry Pseudo- + Packet, as illustrated in Figure 8: + + The secret key and the nonce are values derived by calling HKDF- + Expand-Label using + 0xd9c9943e6101fd200021506bcc02814c73030f25c79d71ce876eca876e6fca8e as + the secret, with labels being "quic key" and "quic iv" (Section 5.1). + + Retry Pseudo-Packet { + ODCID Length (8), + Original Destination Connection ID (0..160), + Header Form (1) = 1, + Fixed Bit (1) = 1, + Long Packet Type (2) = 3, + Unused (4), + Version (32), + DCID Len (8), + Destination Connection ID (0..160), + SCID Len (8), + Source Connection ID (0..160), + Retry Token (..), + } + + Figure 8: Retry Pseudo-Packet + + The Retry Pseudo-Packet is not sent over the wire. It is computed by + taking the transmitted Retry packet, removing the Retry Integrity + Tag, and prepending the two following fields: + + ODCID Length: The ODCID Length field contains the length in bytes of + the Original Destination Connection ID field that follows it, + encoded as an 8-bit unsigned integer. + + Original Destination Connection ID: The Original Destination + Connection ID contains the value of the Destination Connection ID + from the Initial packet that this Retry is in response to. The + length of this field is given in ODCID Length. The presence of + this field ensures that a valid Retry packet can only be sent by + an entity that observes the Initial packet. + +6. Key Update + + Once the handshake is confirmed (see Section 4.1.2), an endpoint MAY + initiate a key update. + + The Key Phase bit indicates which packet protection keys are used to + protect the packet. The Key Phase bit is initially set to 0 for the + first set of 1-RTT packets and toggled to signal each subsequent key + update. + + The Key Phase bit allows a recipient to detect a change in keying + material without needing to receive the first packet that triggered + the change. An endpoint that notices a changed Key Phase bit updates + keys and decrypts the packet that contains the changed value. + + Initiating a key update results in both endpoints updating keys. + This differs from TLS where endpoints can update keys independently. + + This mechanism replaces the key update mechanism of TLS, which relies + on KeyUpdate messages sent using 1-RTT encryption keys. Endpoints + MUST NOT send a TLS KeyUpdate message. Endpoints MUST treat the + receipt of a TLS KeyUpdate message as a connection error of type + 0x010a, equivalent to a fatal TLS alert of unexpected_message; see + Section 4.8. + + Figure 9 shows a key update process, where the initial set of keys + used (identified with @M) are replaced by updated keys (identified + with @N). The value of the Key Phase bit is indicated in brackets + []. + + Initiating Peer Responding Peer + + @M [0] QUIC Packets + + ... Update to @N + @N [1] QUIC Packets + --------> + Update to @N ... + QUIC Packets [1] @N + <-------- + QUIC Packets [1] @N + containing ACK + <-------- + ... Key Update Permitted + + @N [1] QUIC Packets + containing ACK for @N packets + --------> + Key Update Permitted ... + + Figure 9: Key Update + +6.1. Initiating a Key Update + + Endpoints maintain separate read and write secrets for packet + protection. An endpoint initiates a key update by updating its + packet protection write secret and using that to protect new packets. + The endpoint creates a new write secret from the existing write + secret as performed in Section 7.2 of [TLS13]. This uses the KDF + function provided by TLS with a label of "quic ku". The + corresponding key and IV are created from that secret as defined in + Section 5.1. The header protection key is not updated. + + For example, to update write keys with TLS 1.3, HKDF-Expand-Label is + used as: + + secret_<n+1> = HKDF-Expand-Label(secret_<n>, "quic ku", + "", Hash.length) + + The endpoint toggles the value of the Key Phase bit and uses the + updated key and IV to protect all subsequent packets. + + An endpoint MUST NOT initiate a key update prior to having confirmed + the handshake (Section 4.1.2). An endpoint MUST NOT initiate a + subsequent key update unless it has received an acknowledgment for a + packet that was sent protected with keys from the current key phase. + This ensures that keys are available to both peers before another key + update can be initiated. This can be implemented by tracking the + lowest packet number sent with each key phase and the highest + acknowledged packet number in the 1-RTT space: once the latter is + higher than or equal to the former, another key update can be + initiated. + + | Note: Keys of packets other than the 1-RTT packets are never + | updated; their keys are derived solely from the TLS handshake + | state. + + The endpoint that initiates a key update also updates the keys that + it uses for receiving packets. These keys will be needed to process + packets the peer sends after updating. + + An endpoint MUST retain old keys until it has successfully + unprotected a packet sent using the new keys. An endpoint SHOULD + retain old keys for some time after unprotecting a packet sent using + the new keys. Discarding old keys too early can cause delayed + packets to be discarded. Discarding packets will be interpreted as + packet loss by the peer and could adversely affect performance. + +6.2. Responding to a Key Update + + A peer is permitted to initiate a key update after receiving an + acknowledgment of a packet in the current key phase. An endpoint + detects a key update when processing a packet with a key phase that + differs from the value used to protect the last packet it sent. To + process this packet, the endpoint uses the next packet protection key + and IV. See Section 6.3 for considerations about generating these + keys. + + If a packet is successfully processed using the next key and IV, then + the peer has initiated a key update. The endpoint MUST update its + send keys to the corresponding key phase in response, as described in + Section 6.1. Sending keys MUST be updated before sending an + acknowledgment for the packet that was received with updated keys. + By acknowledging the packet that triggered the key update in a packet + protected with the updated keys, the endpoint signals that the key + update is complete. + + An endpoint can defer sending the packet or acknowledgment according + to its normal packet sending behavior; it is not necessary to + immediately generate a packet in response to a key update. The next + packet sent by the endpoint will use the updated keys. The next + packet that contains an acknowledgment will cause the key update to + be completed. If an endpoint detects a second update before it has + sent any packets with updated keys containing an acknowledgment for + the packet that initiated the key update, it indicates that its peer + has updated keys twice without awaiting confirmation. An endpoint + MAY treat such consecutive key updates as a connection error of type + KEY_UPDATE_ERROR. + + An endpoint that receives an acknowledgment that is carried in a + packet protected with old keys where any acknowledged packet was + protected with newer keys MAY treat that as a connection error of + type KEY_UPDATE_ERROR. This indicates that a peer has received and + acknowledged a packet that initiates a key update, but has not + updated keys in response. + +6.3. Timing of Receive Key Generation + + Endpoints responding to an apparent key update MUST NOT generate a + timing side-channel signal that might indicate that the Key Phase bit + was invalid (see Section 9.5). Endpoints can use randomized packet + protection keys in place of discarded keys when key updates are not + yet permitted. Using randomized keys ensures that attempting to + remove packet protection does not result in timing variations, and + results in packets with an invalid Key Phase bit being rejected. + + The process of creating new packet protection keys for receiving + packets could reveal that a key update has occurred. An endpoint MAY + generate new keys as part of packet processing, but this creates a + timing signal that could be used by an attacker to learn when key + updates happen and thus leak the value of the Key Phase bit. + + Endpoints are generally expected to have current and next receive + packet protection keys available. For a short period after a key + update completes, up to the PTO, endpoints MAY defer generation of + the next set of receive packet protection keys. This allows + endpoints to retain only two sets of receive keys; see Section 6.5. + + Once generated, the next set of packet protection keys SHOULD be + retained, even if the packet that was received was subsequently + discarded. Packets containing apparent key updates are easy to + forge, and while the process of key update does not require + significant effort, triggering this process could be used by an + attacker for DoS. + + For this reason, endpoints MUST be able to retain two sets of packet + protection keys for receiving packets: the current and the next. + Retaining the previous keys in addition to these might improve + performance, but this is not essential. + +6.4. Sending with Updated Keys + + An endpoint never sends packets that are protected with old keys. + Only the current keys are used. Keys used for protecting packets can + be discarded immediately after switching to newer keys. + + Packets with higher packet numbers MUST be protected with either the + same or newer packet protection keys than packets with lower packet + numbers. An endpoint that successfully removes protection with old + keys when newer keys were used for packets with lower packet numbers + MUST treat this as a connection error of type KEY_UPDATE_ERROR. + +6.5. Receiving with Different Keys + + For receiving packets during a key update, packets protected with + older keys might arrive if they were delayed by the network. + Retaining old packet protection keys allows these packets to be + successfully processed. + + As packets protected with keys from the next key phase use the same + Key Phase value as those protected with keys from the previous key + phase, it is necessary to distinguish between the two if packets + protected with old keys are to be processed. This can be done using + packet numbers. A recovered packet number that is lower than any + packet number from the current key phase uses the previous packet + protection keys; a recovered packet number that is higher than any + packet number from the current key phase requires the use of the next + packet protection keys. + + Some care is necessary to ensure that any process for selecting + between previous, current, and next packet protection keys does not + expose a timing side channel that might reveal which keys were used + to remove packet protection. See Section 9.5 for more information. + + Alternatively, endpoints can retain only two sets of packet + protection keys, swapping previous for next after enough time has + passed to allow for reordering in the network. In this case, the Key + Phase bit alone can be used to select keys. + + An endpoint MAY allow a period of approximately the Probe Timeout + (PTO; see [QUIC-RECOVERY]) after promoting the next set of receive + keys to be current before it creates the subsequent set of packet + protection keys. These updated keys MAY replace the previous keys at + that time. With the caveat that PTO is a subjective measure -- that + is, a peer could have a different view of the RTT -- this time is + expected to be long enough that any reordered packets would be + declared lost by a peer even if they were acknowledged and short + enough to allow a peer to initiate further key updates. + + Endpoints need to allow for the possibility that a peer might not be + able to decrypt packets that initiate a key update during the period + when the peer retains old keys. Endpoints SHOULD wait three times + the PTO before initiating a key update after receiving an + acknowledgment that confirms that the previous key update was + received. Failing to allow sufficient time could lead to packets + being discarded. + + An endpoint SHOULD retain old read keys for no more than three times + the PTO after having received a packet protected using the new keys. + After this period, old read keys and their corresponding secrets + SHOULD be discarded. + +6.6. Limits on AEAD Usage + + This document sets usage limits for AEAD algorithms to ensure that + overuse does not give an adversary a disproportionate advantage in + attacking the confidentiality and integrity of communications when + using QUIC. + + The usage limits defined in TLS 1.3 exist for protection against + attacks on confidentiality and apply to successful applications of + AEAD protection. The integrity protections in authenticated + encryption also depend on limiting the number of attempts to forge + packets. TLS achieves this by closing connections after any record + fails an authentication check. In comparison, QUIC ignores any + packet that cannot be authenticated, allowing multiple forgery + attempts. + + QUIC accounts for AEAD confidentiality and integrity limits + separately. The confidentiality limit applies to the number of + packets encrypted with a given key. The integrity limit applies to + the number of packets decrypted within a given connection. Details + on enforcing these limits for each AEAD algorithm follow below. + + Endpoints MUST count the number of encrypted packets for each set of + keys. If the total number of encrypted packets with the same key + exceeds the confidentiality limit for the selected AEAD, the endpoint + MUST stop using those keys. Endpoints MUST initiate a key update + before sending more protected packets than the confidentiality limit + for the selected AEAD permits. If a key update is not possible or + integrity limits are reached, the endpoint MUST stop using the + connection and only send stateless resets in response to receiving + packets. It is RECOMMENDED that endpoints immediately close the + connection with a connection error of type AEAD_LIMIT_REACHED before + reaching a state where key updates are not possible. + + For AEAD_AES_128_GCM and AEAD_AES_256_GCM, the confidentiality limit + is 2^23 encrypted packets; see Appendix B.1. For + AEAD_CHACHA20_POLY1305, the confidentiality limit is greater than the + number of possible packets (2^62) and so can be disregarded. For + AEAD_AES_128_CCM, the confidentiality limit is 2^21.5 encrypted + packets; see Appendix B.2. Applying a limit reduces the probability + that an attacker can distinguish the AEAD in use from a random + permutation; see [AEBounds], [ROBUST], and [GCM-MU]. + + In addition to counting packets sent, endpoints MUST count the number + of received packets that fail authentication during the lifetime of a + connection. If the total number of received packets that fail + authentication within the connection, across all keys, exceeds the + integrity limit for the selected AEAD, the endpoint MUST immediately + close the connection with a connection error of type + AEAD_LIMIT_REACHED and not process any more packets. + + For AEAD_AES_128_GCM and AEAD_AES_256_GCM, the integrity limit is + 2^52 invalid packets; see Appendix B.1. For AEAD_CHACHA20_POLY1305, + the integrity limit is 2^36 invalid packets; see [AEBounds]. For + AEAD_AES_128_CCM, the integrity limit is 2^21.5 invalid packets; see + Appendix B.2. Applying this limit reduces the probability that an + attacker can successfully forge a packet; see [AEBounds], [ROBUST], + and [GCM-MU]. + + Endpoints that limit the size of packets MAY use higher + confidentiality and integrity limits; see Appendix B for details. + + Future analyses and specifications MAY relax confidentiality or + integrity limits for an AEAD. + + Any TLS cipher suite that is specified for use with QUIC MUST define + limits on the use of the associated AEAD function that preserves + margins for confidentiality and integrity. That is, limits MUST be + specified for the number of packets that can be authenticated and for + the number of packets that can fail authentication. Providing a + reference to any analysis upon which values are based -- and any + assumptions used in that analysis -- allows limits to be adapted to + varying usage conditions. + +6.7. Key Update Error Code + + The KEY_UPDATE_ERROR error code (0x0e) is used to signal errors + related to key updates. + +7. Security of Initial Messages + + Initial packets are not protected with a secret key, so they are + subject to potential tampering by an attacker. QUIC provides + protection against attackers that cannot read packets but does not + attempt to provide additional protection against attacks where the + attacker can observe and inject packets. Some forms of tampering -- + such as modifying the TLS messages themselves -- are detectable, but + some -- such as modifying ACKs -- are not. + + For example, an attacker could inject a packet containing an ACK + frame to make it appear that a packet had not been received or to + create a false impression of the state of the connection (e.g., by + modifying the ACK Delay). Note that such a packet could cause a + legitimate packet to be dropped as a duplicate. Implementations + SHOULD use caution in relying on any data that is contained in + Initial packets that is not otherwise authenticated. + + It is also possible for the attacker to tamper with data that is + carried in Handshake packets, but because that sort of tampering + requires modifying TLS handshake messages, any such tampering will + cause the TLS handshake to fail. + +8. QUIC-Specific Adjustments to the TLS Handshake + + Certain aspects of the TLS handshake are different when used with + QUIC. + + QUIC also requires additional features from TLS. In addition to + negotiation of cryptographic parameters, the TLS handshake carries + and authenticates values for QUIC transport parameters. + +8.1. Protocol Negotiation + + QUIC requires that the cryptographic handshake provide authenticated + protocol negotiation. TLS uses Application-Layer Protocol + Negotiation [ALPN] to select an application protocol. Unless another + mechanism is used for agreeing on an application protocol, endpoints + MUST use ALPN for this purpose. + + When using ALPN, endpoints MUST immediately close a connection (see + Section 10.2 of [QUIC-TRANSPORT]) with a no_application_protocol TLS + alert (QUIC error code 0x0178; see Section 4.8) if an application + protocol is not negotiated. While [ALPN] only specifies that servers + use this alert, QUIC clients MUST use error 0x0178 to terminate a + connection when ALPN negotiation fails. + + An application protocol MAY restrict the QUIC versions that it can + operate over. Servers MUST select an application protocol compatible + with the QUIC version that the client has selected. The server MUST + treat the inability to select a compatible application protocol as a + connection error of type 0x0178 (no_application_protocol). + Similarly, a client MUST treat the selection of an incompatible + application protocol by a server as a connection error of type + 0x0178. + +8.2. QUIC Transport Parameters Extension + + QUIC transport parameters are carried in a TLS extension. Different + versions of QUIC might define a different method for negotiating + transport configuration. + + Including transport parameters in the TLS handshake provides + integrity protection for these values. + + enum { + quic_transport_parameters(0x39), (65535) + } ExtensionType; + + The extension_data field of the quic_transport_parameters extension + contains a value that is defined by the version of QUIC that is in + use. + + The quic_transport_parameters extension is carried in the ClientHello + and the EncryptedExtensions messages during the handshake. Endpoints + MUST send the quic_transport_parameters extension; endpoints that + receive ClientHello or EncryptedExtensions messages without the + quic_transport_parameters extension MUST close the connection with an + error of type 0x016d (equivalent to a fatal TLS missing_extension + alert, see Section 4.8). + + Transport parameters become available prior to the completion of the + handshake. A server might use these values earlier than handshake + completion. However, the value of transport parameters is not + authenticated until the handshake completes, so any use of these + parameters cannot depend on their authenticity. Any tampering with + transport parameters will cause the handshake to fail. + + Endpoints MUST NOT send this extension in a TLS connection that does + not use QUIC (such as the use of TLS with TCP defined in [TLS13]). A + fatal unsupported_extension alert MUST be sent by an implementation + that supports this extension if the extension is received when the + transport is not QUIC. + + Negotiating the quic_transport_parameters extension causes the + EndOfEarlyData to be removed; see Section 8.3. + +8.3. Removing the EndOfEarlyData Message + + The TLS EndOfEarlyData message is not used with QUIC. QUIC does not + rely on this message to mark the end of 0-RTT data or to signal the + change to Handshake keys. + + Clients MUST NOT send the EndOfEarlyData message. A server MUST + treat receipt of a CRYPTO frame in a 0-RTT packet as a connection + error of type PROTOCOL_VIOLATION. + + As a result, EndOfEarlyData does not appear in the TLS handshake + transcript. + +8.4. Prohibit TLS Middlebox Compatibility Mode + + Appendix D.4 of [TLS13] describes an alteration to the TLS 1.3 + handshake as a workaround for bugs in some middleboxes. The TLS 1.3 + middlebox compatibility mode involves setting the legacy_session_id + field to a 32-byte value in the ClientHello and ServerHello, then + sending a change_cipher_spec record. Both field and record carry no + semantic content and are ignored. + + This mode has no use in QUIC as it only applies to middleboxes that + interfere with TLS over TCP. QUIC also provides no means to carry a + change_cipher_spec record. A client MUST NOT request the use of the + TLS 1.3 compatibility mode. A server SHOULD treat the receipt of a + TLS ClientHello with a non-empty legacy_session_id field as a + connection error of type PROTOCOL_VIOLATION. + +9. Security Considerations + + All of the security considerations that apply to TLS also apply to + the use of TLS in QUIC. Reading all of [TLS13] and its appendices is + the best way to gain an understanding of the security properties of + QUIC. + + This section summarizes some of the more important security aspects + specific to the TLS integration, though there are many security- + relevant details in the remainder of the document. + +9.1. Session Linkability + + Use of TLS session tickets allows servers and possibly other entities + to correlate connections made by the same client; see Section 4.5 for + details. + +9.2. Replay Attacks with 0-RTT + + As described in Section 8 of [TLS13], use of TLS early data comes + with an exposure to replay attack. The use of 0-RTT in QUIC is + similarly vulnerable to replay attack. + + Endpoints MUST implement and use the replay protections described in + [TLS13], however it is recognized that these protections are + imperfect. Therefore, additional consideration of the risk of replay + is needed. + + QUIC is not vulnerable to replay attack, except via the application + protocol information it might carry. The management of QUIC protocol + state based on the frame types defined in [QUIC-TRANSPORT] is not + vulnerable to replay. Processing of QUIC frames is idempotent and + cannot result in invalid connection states if frames are replayed, + reordered, or lost. QUIC connections do not produce effects that + last beyond the lifetime of the connection, except for those produced + by the application protocol that QUIC serves. + + TLS session tickets and address validation tokens are used to carry + QUIC configuration information between connections, specifically, to + enable a server to efficiently recover state that is used in + connection establishment and address validation. These MUST NOT be + used to communicate application semantics between endpoints; clients + MUST treat them as opaque values. The potential for reuse of these + tokens means that they require stronger protections against replay. + + A server that accepts 0-RTT on a connection incurs a higher cost than + accepting a connection without 0-RTT. This includes higher + processing and computation costs. Servers need to consider the + probability of replay and all associated costs when accepting 0-RTT. + + Ultimately, the responsibility for managing the risks of replay + attacks with 0-RTT lies with an application protocol. An application + protocol that uses QUIC MUST describe how the protocol uses 0-RTT and + the measures that are employed to protect against replay attack. An + analysis of replay risk needs to consider all QUIC protocol features + that carry application semantics. + + Disabling 0-RTT entirely is the most effective defense against replay + attack. + + QUIC extensions MUST either describe how replay attacks affect their + operation or prohibit the use of the extension in 0-RTT. Application + protocols MUST either prohibit the use of extensions that carry + application semantics in 0-RTT or provide replay mitigation + strategies. + +9.3. Packet Reflection Attack Mitigation + + A small ClientHello that results in a large block of handshake + messages from a server can be used in packet reflection attacks to + amplify the traffic generated by an attacker. + + QUIC includes three defenses against this attack. First, the packet + containing a ClientHello MUST be padded to a minimum size. Second, + if responding to an unverified source address, the server is + forbidden to send more than three times as many bytes as the number + of bytes it has received (see Section 8.1 of [QUIC-TRANSPORT]). + Finally, because acknowledgments of Handshake packets are + authenticated, a blind attacker cannot forge them. Put together, + these defenses limit the level of amplification. + +9.4. Header Protection Analysis + + [NAN] analyzes authenticated encryption algorithms that provide nonce + privacy, referred to as "Hide Nonce" (HN) transforms. The general + header protection construction in this document is one of those + algorithms (HN1). Header protection is applied after the packet + protection AEAD, sampling a set of bytes ("sample") from the AEAD + output and encrypting the header field using a pseudorandom function + (PRF) as follows: + + protected_field = field XOR PRF(hp_key, sample) + + The header protection variants in this document use a pseudorandom + permutation (PRP) in place of a generic PRF. However, since all PRPs + are also PRFs [IMC], these variants do not deviate from the HN1 + construction. + + As "hp_key" is distinct from the packet protection key, it follows + that header protection achieves AE2 security as defined in [NAN] and + therefore guarantees privacy of "field", the protected packet header. + Future header protection variants based on this construction MUST use + a PRF to ensure equivalent security guarantees. + + Use of the same key and ciphertext sample more than once risks + compromising header protection. Protecting two different headers + with the same key and ciphertext sample reveals the exclusive OR of + the protected fields. Assuming that the AEAD acts as a PRF, if L + bits are sampled, the odds of two ciphertext samples being identical + approach 2^(-L/2), that is, the birthday bound. For the algorithms + described in this document, that probability is one in 2^64. + + To prevent an attacker from modifying packet headers, the header is + transitively authenticated using packet protection; the entire packet + header is part of the authenticated additional data. Protected + fields that are falsified or modified can only be detected once the + packet protection is removed. + +9.5. Header Protection Timing Side Channels + + An attacker could guess values for packet numbers or Key Phase and + have an endpoint confirm guesses through timing side channels. + Similarly, guesses for the packet number length can be tried and + exposed. If the recipient of a packet discards packets with + duplicate packet numbers without attempting to remove packet + protection, they could reveal through timing side channels that the + packet number matches a received packet. For authentication to be + free from side channels, the entire process of header protection + removal, packet number recovery, and packet protection removal MUST + be applied together without timing and other side channels. + + For the sending of packets, construction and protection of packet + payloads and packet numbers MUST be free from side channels that + would reveal the packet number or its encoded size. + + During a key update, the time taken to generate new keys could reveal + through timing side channels that a key update has occurred. + Alternatively, where an attacker injects packets, this side channel + could reveal the value of the Key Phase on injected packets. After + receiving a key update, an endpoint SHOULD generate and save the next + set of receive packet protection keys, as described in Section 6.3. + By generating new keys before a key update is received, receipt of + packets will not create timing signals that leak the value of the Key + Phase. + + This depends on not doing this key generation during packet + processing, and it can require that endpoints maintain three sets of + packet protection keys for receiving: for the previous key phase, for + the current key phase, and for the next key phase. Endpoints can + instead choose to defer generation of the next receive packet + protection keys until they discard old keys so that only two sets of + receive keys need to be retained at any point in time. + +9.6. Key Diversity + + In using TLS, the central key schedule of TLS is used. As a result + of the TLS handshake messages being integrated into the calculation + of secrets, the inclusion of the QUIC transport parameters extension + ensures that the handshake and 1-RTT keys are not the same as those + that might be produced by a server running TLS over TCP. To avoid + the possibility of cross-protocol key synchronization, additional + measures are provided to improve key separation. + + The QUIC packet protection keys and IVs are derived using a different + label than the equivalent keys in TLS. + + To preserve this separation, a new version of QUIC SHOULD define new + labels for key derivation for packet protection key and IV, plus the + header protection keys. This version of QUIC uses the string "quic". + Other versions can use a version-specific label in place of that + string. + + The initial secrets use a key that is specific to the negotiated QUIC + version. New QUIC versions SHOULD define a new salt value used in + calculating initial secrets. + +9.7. Randomness + + QUIC depends on endpoints being able to generate secure random + numbers, both directly for protocol values such as the connection ID, + and transitively via TLS. See [RFC4086] for guidance on secure + random number generation. + +10. IANA Considerations + + IANA has registered a codepoint of 57 (or 0x39) for the + quic_transport_parameters extension (defined in Section 8.2) in the + "TLS ExtensionType Values" registry [TLS-REGISTRIES]. + + The Recommended column for this extension is marked Yes. The TLS 1.3 + Column includes CH (ClientHello) and EE (EncryptedExtensions). + + +=======+===========================+=====+=============+===========+ + | Value | Extension Name | TLS | Recommended | Reference | + | | | 1.3 | | | + +=======+===========================+=====+=============+===========+ + | 57 | quic_transport_parameters | CH, | Y | This | + | | | EE | | document | + +-------+---------------------------+-----+-------------+-----------+ + + Table 2: TLS ExtensionType Values Registry Entry + +11. References + +11.1. Normative References + + [AEAD] McGrew, D., "An Interface and Algorithms for Authenticated + Encryption", RFC 5116, DOI 10.17487/RFC5116, January 2008, + <https://www.rfc-editor.org/info/rfc5116>. + + [AES] "Advanced encryption standard (AES)", National Institute + of Standards and Technology report, + DOI 10.6028/nist.fips.197, November 2001, + <https://doi.org/10.6028/nist.fips.197>. + + [ALPN] Friedl, S., Popov, A., Langley, A., and E. Stephan, + "Transport Layer Security (TLS) Application-Layer Protocol + Negotiation Extension", RFC 7301, DOI 10.17487/RFC7301, + July 2014, <https://www.rfc-editor.org/info/rfc7301>. + + [CHACHA] Nir, Y. and A. Langley, "ChaCha20 and Poly1305 for IETF + Protocols", RFC 8439, DOI 10.17487/RFC8439, June 2018, + <https://www.rfc-editor.org/info/rfc8439>. + + [HKDF] Krawczyk, H. and P. Eronen, "HMAC-based Extract-and-Expand + Key Derivation Function (HKDF)", RFC 5869, + DOI 10.17487/RFC5869, May 2010, + <https://www.rfc-editor.org/info/rfc5869>. + + [QUIC-RECOVERY] + Iyengar, J., Ed. and I. Swett, Ed., "QUIC Loss Detection + and Congestion Control", RFC 9002, DOI 10.17487/RFC9002, + May 2021, <https://www.rfc-editor.org/info/rfc9002>. + + [QUIC-TRANSPORT] + Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC4086] Eastlake 3rd, D., Schiller, J., and S. Crocker, + "Randomness Requirements for Security", BCP 106, RFC 4086, + DOI 10.17487/RFC4086, June 2005, + <https://www.rfc-editor.org/info/rfc4086>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + + [SHA] Dang, Q., "Secure Hash Standard", National Institute of + Standards and Technology report, + DOI 10.6028/nist.fips.180-4, July 2015, + <https://doi.org/10.6028/nist.fips.180-4>. + + [TLS-REGISTRIES] + Salowey, J. and S. Turner, "IANA Registry Updates for TLS + and DTLS", RFC 8447, DOI 10.17487/RFC8447, August 2018, + <https://www.rfc-editor.org/info/rfc8447>. + + [TLS13] Rescorla, E., "The Transport Layer Security (TLS) Protocol + Version 1.3", RFC 8446, DOI 10.17487/RFC8446, August 2018, + <https://www.rfc-editor.org/info/rfc8446>. + +11.2. Informative References + + [AEBounds] Luykx, A. and K. Paterson, "Limits on Authenticated + Encryption Use in TLS", 28 August 2017, + <https://www.isg.rhul.ac.uk/~kp/TLS-AEbounds.pdf>. + + [ASCII] Cerf, V., "ASCII format for network interchange", STD 80, + RFC 20, DOI 10.17487/RFC0020, October 1969, + <https://www.rfc-editor.org/info/rfc20>. + + [CCM-ANALYSIS] + Jonsson, J., "On the Security of CTR + CBC-MAC", Selected + Areas in Cryptography, SAC 2002, Lecture Notes in Computer + Science, vol 2595, pp. 76-93, DOI 10.1007/3-540-36492-7_7, + 2003, <https://doi.org/10.1007/3-540-36492-7_7>. + + [COMPRESS] Ghedini, A. and V. Vasiliev, "TLS Certificate + Compression", RFC 8879, DOI 10.17487/RFC8879, December + 2020, <https://www.rfc-editor.org/info/rfc8879>. + + [GCM-MU] Hoang, V., Tessaro, S., and A. Thiruvengadam, "The Multi- + user Security of GCM, Revisited: Tight Bounds for Nonce + Randomization", CCS '18: Proceedings of the 2018 ACM + SIGSAC Conference on Computer and Communications Security, + pp. 1429-1440, DOI 10.1145/3243734.3243816, 2018, + <https://doi.org/10.1145/3243734.3243816>. + + [HTTP-REPLAY] + Thomson, M., Nottingham, M., and W. Tarreau, "Using Early + Data in HTTP", RFC 8470, DOI 10.17487/RFC8470, September + 2018, <https://www.rfc-editor.org/info/rfc8470>. + + [HTTP2-TLS13] + Benjamin, D., "Using TLS 1.3 with HTTP/2", RFC 8740, + DOI 10.17487/RFC8740, February 2020, + <https://www.rfc-editor.org/info/rfc8740>. + + [IMC] Katz, J. and Y. Lindell, "Introduction to Modern + Cryptography, Second Edition", ISBN 978-1466570269, 6 + November 2014. + + [NAN] Bellare, M., Ng, R., and B. Tackmann, "Nonces Are Noticed: + AEAD Revisited", Advances in Cryptology - CRYPTO 2019, + Lecture Notes in Computer Science, vol 11692, pp. 235-265, + DOI 10.1007/978-3-030-26948-7_9, 2019, + <https://doi.org/10.1007/978-3-030-26948-7_9>. + + [QUIC-HTTP] + Bishop, M., Ed., "Hypertext Transfer Protocol Version 3 + (HTTP/3)", Work in Progress, Internet-Draft, draft-ietf- + quic-http-34, 2 February 2021, + <https://tools.ietf.org/html/draft-ietf-quic-http-34>. + + [RFC2818] Rescorla, E., "HTTP Over TLS", RFC 2818, + DOI 10.17487/RFC2818, May 2000, + <https://www.rfc-editor.org/info/rfc2818>. + + [RFC5280] Cooper, D., Santesson, S., Farrell, S., Boeyen, S., + Housley, R., and W. Polk, "Internet X.509 Public Key + Infrastructure Certificate and Certificate Revocation List + (CRL) Profile", RFC 5280, DOI 10.17487/RFC5280, May 2008, + <https://www.rfc-editor.org/info/rfc5280>. + + [ROBUST] Fischlin, M., Gรผnther, F., and C. Janson, "Robust + Channels: Handling Unreliable Networks in the Record + Layers of QUIC and DTLS 1.3", 16 May 2020, + <https://eprint.iacr.org/2020/718>. + +Appendix A. Sample Packet Protection + + This section shows examples of packet protection so that + implementations can be verified incrementally. Samples of Initial + packets from both client and server plus a Retry packet are defined. + These packets use an 8-byte client-chosen Destination Connection ID + of 0x8394c8f03e515708. Some intermediate values are included. All + values are shown in hexadecimal. + +A.1. Keys + + The labels generated during the execution of the HKDF-Expand-Label + function (that is, HkdfLabel.label) and part of the value given to + the HKDF-Expand function in order to produce its output are: + + client in: 00200f746c73313320636c69656e7420696e00 + + server in: 00200f746c7331332073657276657220696e00 + + quic key: 00100e746c7331332071756963206b657900 + + quic iv: 000c0d746c733133207175696320697600 + + quic hp: 00100d746c733133207175696320687000 + + The initial secret is common: + + initial_secret = HKDF-Extract(initial_salt, cid) + = 7db5df06e7a69e432496adedb0085192 + 3595221596ae2ae9fb8115c1e9ed0a44 + + The secrets for protecting client packets are: + + client_initial_secret + = HKDF-Expand-Label(initial_secret, "client in", "", 32) + = c00cf151ca5be075ed0ebfb5c80323c4 + 2d6b7db67881289af4008f1f6c357aea + + key = HKDF-Expand-Label(client_initial_secret, "quic key", "", 16) + = 1f369613dd76d5467730efcbe3b1a22d + + iv = HKDF-Expand-Label(client_initial_secret, "quic iv", "", 12) + = fa044b2f42a3fd3b46fb255c + + hp = HKDF-Expand-Label(client_initial_secret, "quic hp", "", 16) + = 9f50449e04a0e810283a1e9933adedd2 + + The secrets for protecting server packets are: + + server_initial_secret + = HKDF-Expand-Label(initial_secret, "server in", "", 32) + = 3c199828fd139efd216c155ad844cc81 + fb82fa8d7446fa7d78be803acdda951b + + key = HKDF-Expand-Label(server_initial_secret, "quic key", "", 16) + = cf3a5331653c364c88f0f379b6067e37 + + iv = HKDF-Expand-Label(server_initial_secret, "quic iv", "", 12) + = 0ac1493ca1905853b0bba03e + + hp = HKDF-Expand-Label(server_initial_secret, "quic hp", "", 16) + = c206b8d9b9f0f37644430b490eeaa314 + +A.2. Client Initial + + The client sends an Initial packet. The unprotected payload of this + packet contains the following CRYPTO frame, plus enough PADDING + frames to make a 1162-byte payload: + + 060040f1010000ed0303ebf8fa56f129 39b9584a3896472ec40bb863cfd3e868 + 04fe3a47f06a2b69484c000004130113 02010000c000000010000e00000b6578 + 616d706c652e636f6dff01000100000a 00080006001d00170018001000070005 + 04616c706e0005000501000000000033 00260024001d00209370b2c9caa47fba + baf4559fedba753de171fa71f50f1ce1 5d43e994ec74d748002b000302030400 + 0d0010000e0403050306030203080408 050806002d00020101001c0002400100 + 3900320408ffffffffffffffff050480 00ffff07048000ffff08011001048000 + 75300901100f088394c8f03e51570806 048000ffff + + The unprotected header indicates a length of 1182 bytes: the 4-byte + packet number, 1162 bytes of frames, and the 16-byte authentication + tag. The header includes the connection ID and a packet number of 2: + + c300000001088394c8f03e5157080000449e00000002 + + Protecting the payload produces output that is sampled for header + protection. Because the header uses a 4-byte packet number encoding, + the first 16 bytes of the protected payload is sampled and then + applied to the header as follows: + + sample = d1b1c98dd7689fb8ec11d242b123dc9b + + mask = AES-ECB(hp, sample)[0..4] + = 437b9aec36 + + header[0] ^= mask[0] & 0x0f + = c0 + header[18..21] ^= mask[1..4] + = 7b9aec34 + header = c000000001088394c8f03e5157080000449e7b9aec34 + + The resulting protected packet is: + + c000000001088394c8f03e5157080000 449e7b9aec34d1b1c98dd7689fb8ec11 + d242b123dc9bd8bab936b47d92ec356c 0bab7df5976d27cd449f63300099f399 + 1c260ec4c60d17b31f8429157bb35a12 82a643a8d2262cad67500cadb8e7378c + 8eb7539ec4d4905fed1bee1fc8aafba1 7c750e2c7ace01e6005f80fcb7df6212 + 30c83711b39343fa028cea7f7fb5ff89 eac2308249a02252155e2347b63d58c5 + 457afd84d05dfffdb20392844ae81215 4682e9cf012f9021a6f0be17ddd0c208 + 4dce25ff9b06cde535d0f920a2db1bf3 62c23e596d11a4f5a6cf3948838a3aec + 4e15daf8500a6ef69ec4e3feb6b1d98e 610ac8b7ec3faf6ad760b7bad1db4ba3 + 485e8a94dc250ae3fdb41ed15fb6a8e5 eba0fc3dd60bc8e30c5c4287e53805db + 059ae0648db2f64264ed5e39be2e20d8 2df566da8dd5998ccabdae053060ae6c + 7b4378e846d29f37ed7b4ea9ec5d82e7 961b7f25a9323851f681d582363aa5f8 + 9937f5a67258bf63ad6f1a0b1d96dbd4 faddfcefc5266ba6611722395c906556 + be52afe3f565636ad1b17d508b73d874 3eeb524be22b3dcbc2c7468d54119c74 + 68449a13d8e3b95811a198f3491de3e7 fe942b330407abf82a4ed7c1b311663a + c69890f4157015853d91e923037c227a 33cdd5ec281ca3f79c44546b9d90ca00 + f064c99e3dd97911d39fe9c5d0b23a22 9a234cb36186c4819e8b9c5927726632 + 291d6a418211cc2962e20fe47feb3edf 330f2c603a9d48c0fcb5699dbfe58964 + 25c5bac4aee82e57a85aaf4e2513e4f0 5796b07ba2ee47d80506f8d2c25e50fd + 14de71e6c418559302f939b0e1abd576 f279c4b2e0feb85c1f28ff18f58891ff + ef132eef2fa09346aee33c28eb130ff2 8f5b766953334113211996d20011a198 + e3fc433f9f2541010ae17c1bf202580f 6047472fb36857fe843b19f5984009dd + c324044e847a4f4a0ab34f719595de37 252d6235365e9b84392b061085349d73 + 203a4a13e96f5432ec0fd4a1ee65accd d5e3904df54c1da510b0ff20dcc0c77f + cb2c0e0eb605cb0504db87632cf3d8b4 dae6e705769d1de354270123cb11450e + fc60ac47683d7b8d0f811365565fd98c 4c8eb936bcab8d069fc33bd801b03ade + a2e1fbc5aa463d08ca19896d2bf59a07 1b851e6c239052172f296bfb5e724047 + 90a2181014f3b94a4e97d117b4381303 68cc39dbb2d198065ae3986547926cd2 + 162f40a29f0c3c8745c0f50fba3852e5 66d44575c29d39a03f0cda721984b6f4 + 40591f355e12d439ff150aab7613499d bd49adabc8676eef023b15b65bfc5ca0 + 6948109f23f350db82123535eb8a7433 bdabcb909271a6ecbcb58b936a88cd4e + 8f2e6ff5800175f113253d8fa9ca8885 c2f552e657dc603f252e1a8e308f76f0 + be79e2fb8f5d5fbbe2e30ecadd220723 c8c0aea8078cdfcb3868263ff8f09400 + 54da48781893a7e49ad5aff4af300cd8 04a6b6279ab3ff3afb64491c85194aab + 760d58a606654f9f4400e8b38591356f bf6425aca26dc85244259ff2b19c41b9 + f96f3ca9ec1dde434da7d2d392b905dd f3d1f9af93d1af5950bd493f5aa731b4 + 056df31bd267b6b90a079831aaf579be 0a39013137aac6d404f518cfd4684064 + 7e78bfe706ca4cf5e9c5453e9f7cfd2b 8b4c8d169a44e55c88d4a9a7f9474241 + e221af44860018ab0856972e194cd934 + +A.3. Server Initial + + The server sends the following payload in response, including an ACK + frame, a CRYPTO frame, and no PADDING frames: + + 02000000000600405a020000560303ee fce7f7b37ba1d1632e96677825ddf739 + 88cfc79825df566dc5430b9a045a1200 130100002e00330024001d00209d3c94 + 0d89690b84d08a60993c144eca684d10 81287c834d5311bcf32bb9da1a002b00 + 020304 + + The header from the server includes a new connection ID and a 2-byte + packet number encoding for a packet number of 1: + + c1000000010008f067a5502a4262b50040750001 + + As a result, after protection, the header protection sample is taken + starting from the third protected byte: + + sample = 2cd0991cd25b0aac406a5816b6394100 + mask = 2ec0d8356a + header = cf000000010008f067a5502a4262b5004075c0d9 + + The final protected packet is then: + + cf000000010008f067a5502a4262b500 4075c0d95a482cd0991cd25b0aac406a + 5816b6394100f37a1c69797554780bb3 8cc5a99f5ede4cf73c3ec2493a1839b3 + dbcba3f6ea46c5b7684df3548e7ddeb9 c3bf9c73cc3f3bded74b562bfb19fb84 + 022f8ef4cdd93795d77d06edbb7aaf2f 58891850abbdca3d20398c276456cbc4 + 2158407dd074ee + +A.4. Retry + + This shows a Retry packet that might be sent in response to the + Initial packet in Appendix A.2. The integrity check includes the + client-chosen connection ID value of 0x8394c8f03e515708, but that + value is not included in the final Retry packet: + + ff000000010008f067a5502a4262b574 6f6b656e04a265ba2eff4d829058fb3f + 0f2496ba + +A.5. ChaCha20-Poly1305 Short Header Packet + + This example shows some of the steps required to protect a packet + with a short header. This example uses AEAD_CHACHA20_POLY1305. + + In this example, TLS produces an application write secret from which + a server uses HKDF-Expand-Label to produce four values: a key, an IV, + a header protection key, and the secret that will be used after keys + are updated (this last value is not used further in this example). + + secret + = 9ac312a7f877468ebe69422748ad00a1 + 5443f18203a07d6060f688f30f21632b + + key = HKDF-Expand-Label(secret, "quic key", "", 32) + = c6d98ff3441c3fe1b2182094f69caa2e + d4b716b65488960a7a984979fb23e1c8 + + iv = HKDF-Expand-Label(secret, "quic iv", "", 12) + = e0459b3474bdd0e44a41c144 + + hp = HKDF-Expand-Label(secret, "quic hp", "", 32) + = 25a282b9e82f06f21f488917a4fc8f1b + 73573685608597d0efcb076b0ab7a7a4 + + ku = HKDF-Expand-Label(secret, "quic ku", "", 32) + = 1223504755036d556342ee9361d25342 + 1a826c9ecdf3c7148684b36b714881f9 + + The following shows the steps involved in protecting a minimal packet + with an empty Destination Connection ID. This packet contains a + single PING frame (that is, a payload of just 0x01) and has a packet + number of 654360564. In this example, using a packet number of + length 3 (that is, 49140 is encoded) avoids having to pad the payload + of the packet; PADDING frames would be needed if the packet number is + encoded on fewer bytes. + + pn = 654360564 (decimal) + nonce = e0459b3474bdd0e46d417eb0 + unprotected header = 4200bff4 + payload plaintext = 01 + payload ciphertext = 655e5cd55c41f69080575d7999c25a5bfb + + The resulting ciphertext is the minimum size possible. One byte is + skipped to produce the sample for header protection. + + sample = 5e5cd55c41f69080575d7999c25a5bfb + mask = aefefe7d03 + header = 4cfe4189 + + The protected packet is the smallest possible packet size of 21 + bytes. + + packet = 4cfe4189655e5cd55c41f69080575d7999c25a5bfb + +Appendix B. AEAD Algorithm Analysis + + This section documents analyses used in deriving AEAD algorithm + limits for AEAD_AES_128_GCM, AEAD_AES_128_CCM, and AEAD_AES_256_GCM. + The analyses that follow use symbols for multiplication (*), division + (/), and exponentiation (^), plus parentheses for establishing + precedence. The following symbols are also used: + + t: The size of the authentication tag in bits. For these ciphers, t + is 128. + + n: The size of the block function in bits. For these ciphers, n is + 128. + + k: The size of the key in bits. This is 128 for AEAD_AES_128_GCM + and AEAD_AES_128_CCM; 256 for AEAD_AES_256_GCM. + + l: The number of blocks in each packet (see below). + + q: The number of genuine packets created and protected by endpoints. + This value is the bound on the number of packets that can be + protected before updating keys. + + v: The number of forged packets that endpoints will accept. This + value is the bound on the number of forged packets that an + endpoint can reject before updating keys. + + o: The amount of offline ideal cipher queries made by an adversary. + + The analyses that follow rely on a count of the number of block + operations involved in producing each message. This analysis is + performed for packets of size up to 2^11 (l = 2^7) and 2^16 (l = + 2^12). A size of 2^11 is expected to be a limit that matches common + deployment patterns, whereas the 2^16 is the maximum possible size of + a QUIC packet. Only endpoints that strictly limit packet size can + use the larger confidentiality and integrity limits that are derived + using the smaller packet size. + + For AEAD_AES_128_GCM and AEAD_AES_256_GCM, the message length (l) is + the length of the associated data in blocks plus the length of the + plaintext in blocks. + + For AEAD_AES_128_CCM, the total number of block cipher operations is + the sum of the following: the length of the associated data in + blocks, the length of the ciphertext in blocks, the length of the + plaintext in blocks, plus 1. In this analysis, this is simplified to + a value of twice the length of the packet in blocks (that is, "2l = + 2^8" for packets that are limited to 2^11 bytes, or "2l = 2^13" + otherwise). This simplification is based on the packet containing + all of the associated data and ciphertext. This results in a one to + three block overestimation of the number of operations per packet. + +B.1. Analysis of AEAD_AES_128_GCM and AEAD_AES_256_GCM Usage Limits + + [GCM-MU] specifies concrete bounds for AEAD_AES_128_GCM and + AEAD_AES_256_GCM as used in TLS 1.3 and QUIC. This section documents + this analysis using several simplifying assumptions: + + * The number of ciphertext blocks an attacker uses in forgery + attempts is bounded by v * l, which is the number of forgery + attempts multiplied by the size of each packet (in blocks). + + * The amount of offline work done by an attacker does not dominate + other factors in the analysis. + + The bounds in [GCM-MU] are tighter and more complete than those used + in [AEBounds], which allows for larger limits than those described in + [TLS13]. + +B.1.1. Confidentiality Limit + + For confidentiality, Theorem (4.3) in [GCM-MU] establishes that, for + a single user that does not repeat nonces, the dominant term in + determining the distinguishing advantage between a real and random + AEAD algorithm gained by an attacker is: + + 2 * (q * l)^2 / 2^n + + For a target advantage of 2^-57, this results in the relation: + + q <= 2^35 / l + + Thus, endpoints that do not send packets larger than 2^11 bytes + cannot protect more than 2^28 packets in a single connection without + causing an attacker to gain a more significant advantage than the + target of 2^-57. The limit for endpoints that allow for the packet + size to be as large as 2^16 is instead 2^23. + +B.1.2. Integrity Limit + + For integrity, Theorem (4.3) in [GCM-MU] establishes that an attacker + gains an advantage in successfully forging a packet of no more than + the following: + + (1 / 2^(8 * n)) + ((2 * v) / 2^(2 * n)) + + ((2 * o * v) / 2^(k + n)) + (n * (v + (v * l)) / 2^k) + + The goal is to limit this advantage to 2^-57. For AEAD_AES_128_GCM, + the fourth term in this inequality dominates the rest, so the others + can be removed without significant effect on the result. This + produces the following approximation: + + v <= 2^64 / l + + Endpoints that do not attempt to remove protection from packets + larger than 2^11 bytes can attempt to remove protection from at most + 2^57 packets. Endpoints that do not restrict the size of processed + packets can attempt to remove protection from at most 2^52 packets. + + For AEAD_AES_256_GCM, the same term dominates, but the larger value + of k produces the following approximation: + + v <= 2^192 / l + + This is substantially larger than the limit for AEAD_AES_128_GCM. + However, this document recommends that the same limit be applied to + both functions as either limit is acceptably large. + +B.2. Analysis of AEAD_AES_128_CCM Usage Limits + + TLS [TLS13] and [AEBounds] do not specify limits on usage for + AEAD_AES_128_CCM. However, any AEAD that is used with QUIC requires + limits on use that ensure that both confidentiality and integrity are + preserved. This section documents that analysis. + + [CCM-ANALYSIS] is used as the basis of this analysis. The results of + that analysis are used to derive usage limits that are based on those + chosen in [TLS13]. + + For confidentiality, Theorem 2 in [CCM-ANALYSIS] establishes that an + attacker gains a distinguishing advantage over an ideal pseudorandom + permutation (PRP) of no more than the following: + + (2l * q)^2 / 2^n + + The integrity limit in Theorem 1 in [CCM-ANALYSIS] provides an + attacker a strictly higher advantage for the same number of messages. + As the targets for the confidentiality advantage and the integrity + advantage are the same, only Theorem 1 needs to be considered. + + Theorem 1 establishes that an attacker gains an advantage over an + ideal PRP of no more than the following: + + v / 2^t + (2l * (v + q))^2 / 2^n + + As "t" and "n" are both 128, the first term is negligible relative to + the second, so that term can be removed without a significant effect + on the result. + + This produces a relation that combines both encryption and decryption + attempts with the same limit as that produced by the theorem for + confidentiality alone. For a target advantage of 2^-57, this results + in the following: + + v + q <= 2^34.5 / l + + By setting "q = v", values for both confidentiality and integrity + limits can be produced. Endpoints that limit packets to 2^11 bytes + therefore have both confidentiality and integrity limits of 2^26.5 + packets. Endpoints that do not restrict packet size have a limit of + 2^21.5. + +Contributors + + The IETF QUIC Working Group received an enormous amount of support + from many people. The following people provided substantive + contributions to this document: + + * Adam Langley + * Alessandro Ghedini + * Christian Huitema + * Christopher Wood + * David Schinazi + * Dragana Damjanovic + * Eric Rescorla + * Felix Gรผnther + * Ian Swett + * Jana Iyengar + * ๅฅฅ ไธ€็ฉ‚ (Kazuho Oku) + * Marten Seemann + * Martin Duke + * Mike Bishop + * Mikkel Fahnรธe Jรธrgensen + * Nick Banks + * Nick Harper + * Roberto Peon + * Rui Paulo + * Ryan Hamilton + * Victor Vasiliev + +Authors' Addresses + + Martin Thomson (editor) + Mozilla + + Email: mt@lowentropy.net + + + Sean Turner (editor) + sn3rd + + Email: sean@sn3rd.com diff --git a/eval/corpora/rfc/RFC 9002 - QUIC Loss Detection and Congestion Control.txt b/eval/corpora/rfc/RFC 9002 - QUIC Loss Detection and Congestion Control.txt new file mode 100644 index 00000000..abc7f809 --- /dev/null +++ b/eval/corpora/rfc/RFC 9002 - QUIC Loss Detection and Congestion Control.txt @@ -0,0 +1,2070 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) J. Iyengar, Ed. +Request for Comments: 9002 Fastly +Category: Standards Track I. Swett, Ed. +ISSN: 2070-1721 Google + May 2021 + + + QUIC Loss Detection and Congestion Control + +Abstract + + This document describes loss detection and congestion control + mechanisms for QUIC. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9002. + +Copyright Notice + + Copyright (c) 2021 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Simplified BSD License text as described in Section 4.e of + the Trust Legal Provisions and are provided without warranty as + described in the Simplified BSD License. + +Table of Contents + + 1. Introduction + 2. Conventions and Definitions + 3. Design of the QUIC Transmission Machinery + 4. Relevant Differences between QUIC and TCP + 4.1. Separate Packet Number Spaces + 4.2. Monotonically Increasing Packet Numbers + 4.3. Clearer Loss Epoch + 4.4. No Reneging + 4.5. More ACK Ranges + 4.6. Explicit Correction for Delayed Acknowledgments + 4.7. Probe Timeout Replaces RTO and TLP + 4.8. The Minimum Congestion Window Is Two Packets + 4.9. Handshake Packets Are Not Special + 5. Estimating the Round-Trip Time + 5.1. Generating RTT Samples + 5.2. Estimating min_rtt + 5.3. Estimating smoothed_rtt and rttvar + 6. Loss Detection + 6.1. Acknowledgment-Based Detection + 6.1.1. Packet Threshold + 6.1.2. Time Threshold + 6.2. Probe Timeout + 6.2.1. Computing PTO + 6.2.2. Handshakes and New Paths + 6.2.3. Speeding up Handshake Completion + 6.2.4. Sending Probe Packets + 6.3. Handling Retry Packets + 6.4. Discarding Keys and Packet State + 7. Congestion Control + 7.1. Explicit Congestion Notification + 7.2. Initial and Minimum Congestion Window + 7.3. Congestion Control States + 7.3.1. Slow Start + 7.3.2. Recovery + 7.3.3. Congestion Avoidance + 7.4. Ignoring Loss of Undecryptable Packets + 7.5. Probe Timeout + 7.6. Persistent Congestion + 7.6.1. Duration + 7.6.2. Establishing Persistent Congestion + 7.6.3. Example + 7.7. Pacing + 7.8. Underutilizing the Congestion Window + 8. Security Considerations + 8.1. Loss and Congestion Signals + 8.2. Traffic Analysis + 8.3. Misreporting ECN Markings + 9. References + 9.1. Normative References + 9.2. Informative References + Appendix A. Loss Recovery Pseudocode + A.1. Tracking Sent Packets + A.1.1. Sent Packet Fields + A.2. Constants of Interest + A.3. Variables of Interest + A.4. Initialization + A.5. On Sending a Packet + A.6. On Receiving a Datagram + A.7. On Receiving an Acknowledgment + A.8. Setting the Loss Detection Timer + A.9. On Timeout + A.10. Detecting Lost Packets + A.11. Upon Dropping Initial or Handshake Keys + Appendix B. Congestion Control Pseudocode + B.1. Constants of Interest + B.2. Variables of Interest + B.3. Initialization + B.4. On Packet Sent + B.5. On Packet Acknowledgment + B.6. On New Congestion Event + B.7. Process ECN Information + B.8. On Packets Lost + B.9. Removing Discarded Packets from Bytes in Flight + Contributors + Authors' Addresses + +1. Introduction + + QUIC is a secure, general-purpose transport protocol, described in + [QUIC-TRANSPORT]. This document describes loss detection and + congestion control mechanisms for QUIC. + +2. Conventions and Definitions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in BCP + 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + Definitions of terms that are used in this document: + + Ack-eliciting frames: All frames other than ACK, PADDING, and + CONNECTION_CLOSE are considered ack-eliciting. + + Ack-eliciting packets: Packets that contain ack-eliciting frames + elicit an ACK from the receiver within the maximum acknowledgment + delay and are called ack-eliciting packets. + + In-flight packets: Packets are considered in flight when they are + ack-eliciting or contain a PADDING frame, and they have been sent + but are not acknowledged, declared lost, or discarded along with + old keys. + +3. Design of the QUIC Transmission Machinery + + All transmissions in QUIC are sent with a packet-level header, which + indicates the encryption level and includes a packet sequence number + (referred to below as a packet number). The encryption level + indicates the packet number space, as described in Section 12.3 of + [QUIC-TRANSPORT]. Packet numbers never repeat within a packet number + space for the lifetime of a connection. Packet numbers are sent in + monotonically increasing order within a space, preventing ambiguity. + It is permitted for some packet numbers to never be used, leaving + intentional gaps. + + This design obviates the need for disambiguating between + transmissions and retransmissions; this eliminates significant + complexity from QUIC's interpretation of TCP loss detection + mechanisms. + + QUIC packets can contain multiple frames of different types. The + recovery mechanisms ensure that data and frames that need reliable + delivery are acknowledged or declared lost and sent in new packets as + necessary. The types of frames contained in a packet affect recovery + and congestion control logic: + + * All packets are acknowledged, though packets that contain no ack- + eliciting frames are only acknowledged along with ack-eliciting + packets. + + * Long header packets that contain CRYPTO frames are critical to the + performance of the QUIC handshake and use shorter timers for + acknowledgment. + + * Packets containing frames besides ACK or CONNECTION_CLOSE frames + count toward congestion control limits and are considered to be in + flight. + + * PADDING frames cause packets to contribute toward bytes in flight + without directly causing an acknowledgment to be sent. + +4. Relevant Differences between QUIC and TCP + + Readers familiar with TCP's loss detection and congestion control + will find algorithms here that parallel well-known TCP ones. + However, protocol differences between QUIC and TCP contribute to + algorithmic differences. These protocol differences are briefly + described below. + +4.1. Separate Packet Number Spaces + + QUIC uses separate packet number spaces for each encryption level, + except 0-RTT and all generations of 1-RTT keys use the same packet + number space. Separate packet number spaces ensures that the + acknowledgment of packets sent with one level of encryption will not + cause spurious retransmission of packets sent with a different + encryption level. Congestion control and round-trip time (RTT) + measurement are unified across packet number spaces. + +4.2. Monotonically Increasing Packet Numbers + + TCP conflates transmission order at the sender with delivery order at + the receiver, resulting in the retransmission ambiguity problem + [RETRANSMISSION]. QUIC separates transmission order from delivery + order: packet numbers indicate transmission order, and delivery order + is determined by the stream offsets in STREAM frames. + + QUIC's packet number is strictly increasing within a packet number + space and directly encodes transmission order. A higher packet + number signifies that the packet was sent later, and a lower packet + number signifies that the packet was sent earlier. When a packet + containing ack-eliciting frames is detected lost, QUIC includes + necessary frames in a new packet with a new packet number, removing + ambiguity about which packet is acknowledged when an ACK is received. + Consequently, more accurate RTT measurements can be made, spurious + retransmissions are trivially detected, and mechanisms such as Fast + Retransmit can be applied universally, based only on packet number. + + This design point significantly simplifies loss detection mechanisms + for QUIC. Most TCP mechanisms implicitly attempt to infer + transmission ordering based on TCP sequence numbers -- a nontrivial + task, especially when TCP timestamps are not available. + +4.3. Clearer Loss Epoch + + QUIC starts a loss epoch when a packet is lost. The loss epoch ends + when any packet sent after the start of the epoch is acknowledged. + TCP waits for the gap in the sequence number space to be filled, and + so if a segment is lost multiple times in a row, the loss epoch may + not end for several round trips. Because both should reduce their + congestion windows only once per epoch, QUIC will do it once for + every round trip that experiences loss, while TCP may only do it once + across multiple round trips. + +4.4. No Reneging + + QUIC ACK frames contain information similar to that in TCP Selective + Acknowledgments (SACKs) [RFC2018]. However, QUIC does not allow a + packet acknowledgment to be reneged, greatly simplifying + implementations on both sides and reducing memory pressure on the + sender. + +4.5. More ACK Ranges + + QUIC supports many ACK ranges, as opposed to TCP's three SACK ranges. + In high-loss environments, this speeds recovery, reduces spurious + retransmits, and ensures forward progress without relying on + timeouts. + +4.6. Explicit Correction for Delayed Acknowledgments + + QUIC endpoints measure the delay incurred between when a packet is + received and when the corresponding acknowledgment is sent, allowing + a peer to maintain a more accurate RTT estimate; see Section 13.2 of + [QUIC-TRANSPORT]. + +4.7. Probe Timeout Replaces RTO and TLP + + QUIC uses a probe timeout (PTO; see Section 6.2), with a timer based + on TCP's retransmission timeout (RTO) computation; see [RFC6298]. + QUIC's PTO includes the peer's maximum expected acknowledgment delay + instead of using a fixed minimum timeout. + + Similar to the RACK-TLP loss detection algorithm for TCP [RFC8985], + QUIC does not collapse the congestion window when the PTO expires, + since a single packet loss at the tail does not indicate persistent + congestion. Instead, QUIC collapses the congestion window when + persistent congestion is declared; see Section 7.6. In doing this, + QUIC avoids unnecessary congestion window reductions, obviating the + need for correcting mechanisms such as Forward RTO-Recovery (F-RTO) + [RFC5682]. Since QUIC does not collapse the congestion window on a + PTO expiration, a QUIC sender is not limited from sending more in- + flight packets after a PTO expiration if it still has available + congestion window. This occurs when a sender is application limited + and the PTO timer expires. This is more aggressive than TCP's RTO + mechanism when application limited, but identical when not + application limited. + + QUIC allows probe packets to temporarily exceed the congestion window + whenever the timer expires. + +4.8. The Minimum Congestion Window Is Two Packets + + TCP uses a minimum congestion window of one packet. However, loss of + that single packet means that the sender needs to wait for a PTO to + recover (Section 6.2), which can be much longer than an RTT. Sending + a single ack-eliciting packet also increases the chances of incurring + additional latency when a receiver delays its acknowledgment. + + QUIC therefore recommends that the minimum congestion window be two + packets. While this increases network load, it is considered safe + since the sender will still reduce its sending rate exponentially + under persistent congestion (Section 6.2). + +4.9. Handshake Packets Are Not Special + + TCP treats the loss of SYN or SYN-ACK packet as persistent congestion + and reduces the congestion window to one packet; see [RFC5681]. QUIC + treats loss of a packet containing handshake data the same as other + losses. + +5. Estimating the Round-Trip Time + + At a high level, an endpoint measures the time from when a packet was + sent to when it is acknowledged as an RTT sample. The endpoint uses + RTT samples and peer-reported host delays (see Section 13.2 of + [QUIC-TRANSPORT]) to generate a statistical description of the + network path's RTT. An endpoint computes the following three values + for each path: the minimum value over a period of time (min_rtt), an + exponentially weighted moving average (smoothed_rtt), and the mean + deviation (referred to as "variation" in the rest of this document) + in the observed RTT samples (rttvar). + +5.1. Generating RTT Samples + + An endpoint generates an RTT sample on receiving an ACK frame that + meets the following two conditions: + + * the largest acknowledged packet number is newly acknowledged, and + + * at least one of the newly acknowledged packets was ack-eliciting. + + The RTT sample, latest_rtt, is generated as the time elapsed since + the largest acknowledged packet was sent: + + latest_rtt = ack_time - send_time_of_largest_acked + + An RTT sample is generated using only the largest acknowledged packet + in the received ACK frame. This is because a peer reports + acknowledgment delays for only the largest acknowledged packet in an + ACK frame. While the reported acknowledgment delay is not used by + the RTT sample measurement, it is used to adjust the RTT sample in + subsequent computations of smoothed_rtt and rttvar (Section 5.3). + + To avoid generating multiple RTT samples for a single packet, an ACK + frame SHOULD NOT be used to update RTT estimates if it does not newly + acknowledge the largest acknowledged packet. + + An RTT sample MUST NOT be generated on receiving an ACK frame that + does not newly acknowledge at least one ack-eliciting packet. A peer + usually does not send an ACK frame when only non-ack-eliciting + packets are received. Therefore, an ACK frame that contains + acknowledgments for only non-ack-eliciting packets could include an + arbitrarily large ACK Delay value. Ignoring such ACK frames avoids + complications in subsequent smoothed_rtt and rttvar computations. + + A sender might generate multiple RTT samples per RTT when multiple + ACK frames are received within an RTT. As suggested in [RFC6298], + doing so might result in inadequate history in smoothed_rtt and + rttvar. Ensuring that RTT estimates retain sufficient history is an + open research question. + +5.2. Estimating min_rtt + + min_rtt is the sender's estimate of the minimum RTT observed for a + given network path over a period of time. In this document, min_rtt + is used by loss detection to reject implausibly small RTT samples. + + min_rtt MUST be set to the latest_rtt on the first RTT sample. + min_rtt MUST be set to the lesser of min_rtt and latest_rtt + (Section 5.1) on all other samples. + + An endpoint uses only locally observed times in computing the min_rtt + and does not adjust for acknowledgment delays reported by the peer. + Doing so allows the endpoint to set a lower bound for the + smoothed_rtt based entirely on what it observes (see Section 5.3) and + limits potential underestimation due to erroneously reported delays + by the peer. + + The RTT for a network path may change over time. If a path's actual + RTT decreases, the min_rtt will adapt immediately on the first low + sample. If the path's actual RTT increases, however, the min_rtt + will not adapt to it, allowing future RTT samples that are smaller + than the new RTT to be included in smoothed_rtt. + + Endpoints SHOULD set the min_rtt to the newest RTT sample after + persistent congestion is established. This avoids repeatedly + declaring persistent congestion when the RTT increases. This also + allows a connection to reset its estimate of min_rtt and smoothed_rtt + after a disruptive network event; see Section 5.3. + + Endpoints MAY reestablish the min_rtt at other times in the + connection, such as when traffic volume is low and an acknowledgment + is received with a low acknowledgment delay. Implementations SHOULD + NOT refresh the min_rtt value too often since the actual minimum RTT + of the path is not frequently observable. + +5.3. Estimating smoothed_rtt and rttvar + + smoothed_rtt is an exponentially weighted moving average of an + endpoint's RTT samples, and rttvar estimates the variation in the RTT + samples using a mean variation. + + The calculation of smoothed_rtt uses RTT samples after adjusting them + for acknowledgment delays. These delays are decoded from the ACK + Delay field of ACK frames as described in Section 19.3 of + [QUIC-TRANSPORT]. + + The peer might report acknowledgment delays that are larger than the + peer's max_ack_delay during the handshake (Section 13.2.1 of + [QUIC-TRANSPORT]). To account for this, the endpoint SHOULD ignore + max_ack_delay until the handshake is confirmed, as defined in + Section 4.1.2 of [QUIC-TLS]. When they occur, these large + acknowledgment delays are likely to be non-repeating and limited to + the handshake. The endpoint can therefore use them without limiting + them to the max_ack_delay, avoiding unnecessary inflation of the RTT + estimate. + + Note that a large acknowledgment delay can result in a substantially + inflated smoothed_rtt if there is an error either in the peer's + reporting of the acknowledgment delay or in the endpoint's min_rtt + estimate. Therefore, prior to handshake confirmation, an endpoint + MAY ignore RTT samples if adjusting the RTT sample for acknowledgment + delay causes the sample to be less than the min_rtt. + + After the handshake is confirmed, any acknowledgment delays reported + by the peer that are greater than the peer's max_ack_delay are + attributed to unintentional but potentially repeating delays, such as + scheduler latency at the peer or loss of previous acknowledgments. + Excess delays could also be due to a noncompliant receiver. + Therefore, these extra delays are considered effectively part of path + delay and incorporated into the RTT estimate. + + Therefore, when adjusting an RTT sample using peer-reported + acknowledgment delays, an endpoint: + + * MAY ignore the acknowledgment delay for Initial packets, since + these acknowledgments are not delayed by the peer (Section 13.2.1 + of [QUIC-TRANSPORT]); + + * SHOULD ignore the peer's max_ack_delay until the handshake is + confirmed; + + * MUST use the lesser of the acknowledgment delay and the peer's + max_ack_delay after the handshake is confirmed; and + + * MUST NOT subtract the acknowledgment delay from the RTT sample if + the resulting value is smaller than the min_rtt. This limits the + underestimation of the smoothed_rtt due to a misreporting peer. + + Additionally, an endpoint might postpone the processing of + acknowledgments when the corresponding decryption keys are not + immediately available. For example, a client might receive an + acknowledgment for a 0-RTT packet that it cannot decrypt because + 1-RTT packet protection keys are not yet available to it. In such + cases, an endpoint SHOULD subtract such local delays from its RTT + sample until the handshake is confirmed. + + Similar to [RFC6298], smoothed_rtt and rttvar are computed as + follows. + + An endpoint initializes the RTT estimator during connection + establishment and when the estimator is reset during connection + migration; see Section 9.4 of [QUIC-TRANSPORT]. Before any RTT + samples are available for a new path or when the estimator is reset, + the estimator is initialized using the initial RTT; see + Section 6.2.2. + + smoothed_rtt and rttvar are initialized as follows, where kInitialRtt + contains the initial RTT value: + + smoothed_rtt = kInitialRtt + rttvar = kInitialRtt / 2 + + RTT samples for the network path are recorded in latest_rtt; see + Section 5.1. On the first RTT sample after initialization, the + estimator is reset using that sample. This ensures that the + estimator retains no history of past samples. Packets sent on other + paths do not contribute RTT samples to the current path, as described + in Section 9.4 of [QUIC-TRANSPORT]. + + On the first RTT sample after initialization, smoothed_rtt and rttvar + are set as follows: + + smoothed_rtt = latest_rtt + rttvar = latest_rtt / 2 + + On subsequent RTT samples, smoothed_rtt and rttvar evolve as follows: + + ack_delay = decoded acknowledgment delay from ACK frame + if (handshake confirmed): + ack_delay = min(ack_delay, max_ack_delay) + adjusted_rtt = latest_rtt + if (latest_rtt >= min_rtt + ack_delay): + adjusted_rtt = latest_rtt - ack_delay + smoothed_rtt = 7/8 * smoothed_rtt + 1/8 * adjusted_rtt + rttvar_sample = abs(smoothed_rtt - adjusted_rtt) + rttvar = 3/4 * rttvar + 1/4 * rttvar_sample + +6. Loss Detection + + QUIC senders use acknowledgments to detect lost packets and a PTO to + ensure acknowledgments are received; see Section 6.2. This section + provides a description of these algorithms. + + If a packet is lost, the QUIC transport needs to recover from that + loss, such as by retransmitting the data, sending an updated frame, + or discarding the frame. For more information, see Section 13.3 of + [QUIC-TRANSPORT]. + + Loss detection is separate per packet number space, unlike RTT + measurement and congestion control, because RTT and congestion + control are properties of the path, whereas loss detection also + relies upon key availability. + +6.1. Acknowledgment-Based Detection + + Acknowledgment-based loss detection implements the spirit of TCP's + Fast Retransmit [RFC5681], Early Retransmit [RFC5827], Forward + Acknowledgment [FACK], SACK loss recovery [RFC6675], and RACK-TLP + [RFC8985]. This section provides an overview of how these algorithms + are implemented in QUIC. + + A packet is declared lost if it meets all of the following + conditions: + + * The packet is unacknowledged, in flight, and was sent prior to an + acknowledged packet. + + * The packet was sent kPacketThreshold packets before an + acknowledged packet (Section 6.1.1), or it was sent long enough in + the past (Section 6.1.2). + + The acknowledgment indicates that a packet sent later was delivered, + and the packet and time thresholds provide some tolerance for packet + reordering. + + Spuriously declaring packets as lost leads to unnecessary + retransmissions and may result in degraded performance due to the + actions of the congestion controller upon detecting loss. + Implementations can detect spurious retransmissions and increase the + packet or time reordering threshold to reduce future spurious + retransmissions and loss events. Implementations with adaptive time + thresholds MAY choose to start with smaller initial reordering + thresholds to minimize recovery latency. + +6.1.1. Packet Threshold + + The RECOMMENDED initial value for the packet reordering threshold + (kPacketThreshold) is 3, based on best practices for TCP loss + detection [RFC5681] [RFC6675]. In order to remain similar to TCP, + implementations SHOULD NOT use a packet threshold less than 3; see + [RFC5681]. + + Some networks may exhibit higher degrees of packet reordering, + causing a sender to detect spurious losses. Additionally, packet + reordering could be more common with QUIC than TCP because network + elements that could observe and reorder TCP packets cannot do that + for QUIC and also because QUIC packet numbers are encrypted. + Algorithms that increase the reordering threshold after spuriously + detecting losses, such as RACK [RFC8985], have proven to be useful in + TCP and are expected to be at least as useful in QUIC. + +6.1.2. Time Threshold + + Once a later packet within the same packet number space has been + acknowledged, an endpoint SHOULD declare an earlier packet lost if it + was sent a threshold amount of time in the past. To avoid declaring + packets as lost too early, this time threshold MUST be set to at + least the local timer granularity, as indicated by the kGranularity + constant. The time threshold is: + + max(kTimeThreshold * max(smoothed_rtt, latest_rtt), kGranularity) + + If packets sent prior to the largest acknowledged packet cannot yet + be declared lost, then a timer SHOULD be set for the remaining time. + + Using max(smoothed_rtt, latest_rtt) protects from the two following + cases: + + * the latest RTT sample is lower than the smoothed RTT, perhaps due + to reordering where the acknowledgment encountered a shorter path; + + * the latest RTT sample is higher than the smoothed RTT, perhaps due + to a sustained increase in the actual RTT, but the smoothed RTT + has not yet caught up. + + The RECOMMENDED time threshold (kTimeThreshold), expressed as an RTT + multiplier, is 9/8. The RECOMMENDED value of the timer granularity + (kGranularity) is 1 millisecond. + + | Note: TCP's RACK [RFC8985] specifies a slightly larger + | threshold, equivalent to 5/4, for a similar purpose. + | Experience with QUIC shows that 9/8 works well. + + Implementations MAY experiment with absolute thresholds, thresholds + from previous connections, adaptive thresholds, or the including of + RTT variation. Smaller thresholds reduce reordering resilience and + increase spurious retransmissions, and larger thresholds increase + loss detection delay. + +6.2. Probe Timeout + + A Probe Timeout (PTO) triggers the sending of one or two probe + datagrams when ack-eliciting packets are not acknowledged within the + expected period of time or the server may not have validated the + client's address. A PTO enables a connection to recover from loss of + tail packets or acknowledgments. + + As with loss detection, the PTO is per packet number space. That is, + a PTO value is computed per packet number space. + + A PTO timer expiration event does not indicate packet loss and MUST + NOT cause prior unacknowledged packets to be marked as lost. When an + acknowledgment is received that newly acknowledges packets, loss + detection proceeds as dictated by the packet and time threshold + mechanisms; see Section 6.1. + + The PTO algorithm used in QUIC implements the reliability functions + of Tail Loss Probe [RFC8985], RTO [RFC5681], and F-RTO algorithms for + TCP [RFC5682]. The timeout computation is based on TCP's RTO period + [RFC6298]. + +6.2.1. Computing PTO + + When an ack-eliciting packet is transmitted, the sender schedules a + timer for the PTO period as follows: + + PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay + + The PTO period is the amount of time that a sender ought to wait for + an acknowledgment of a sent packet. This time period includes the + estimated network RTT (smoothed_rtt), the variation in the estimate + (4*rttvar), and max_ack_delay, to account for the maximum time by + which a receiver might delay sending an acknowledgment. + + When the PTO is armed for Initial or Handshake packet number spaces, + the max_ack_delay in the PTO period computation is set to 0, since + the peer is expected to not delay these packets intentionally; see + Section 13.2.1 of [QUIC-TRANSPORT]. + + The PTO period MUST be at least kGranularity to avoid the timer + expiring immediately. + + When ack-eliciting packets in multiple packet number spaces are in + flight, the timer MUST be set to the earlier value of the Initial and + Handshake packet number spaces. + + An endpoint MUST NOT set its PTO timer for the Application Data + packet number space until the handshake is confirmed. Doing so + prevents the endpoint from retransmitting information in packets when + either the peer does not yet have the keys to process them or the + endpoint does not yet have the keys to process their acknowledgments. + For example, this can happen when a client sends 0-RTT packets to the + server; it does so without knowing whether the server will be able to + decrypt them. Similarly, this can happen when a server sends 1-RTT + packets before confirming that the client has verified the server's + certificate and can therefore read these 1-RTT packets. + + A sender SHOULD restart its PTO timer every time an ack-eliciting + packet is sent or acknowledged, or when Initial or Handshake keys are + discarded (Section 4.9 of [QUIC-TLS]). This ensures the PTO is + always set based on the latest estimate of the RTT and for the + correct packet across packet number spaces. + + When a PTO timer expires, the PTO backoff MUST be increased, + resulting in the PTO period being set to twice its current value. + The PTO backoff factor is reset when an acknowledgment is received, + except in the following case. A server might take longer to respond + to packets during the handshake than otherwise. To protect such a + server from repeated client probes, the PTO backoff is not reset at a + client that is not yet certain that the server has finished + validating the client's address. That is, a client does not reset + the PTO backoff factor on receiving acknowledgments in Initial + packets. + + This exponential reduction in the sender's rate is important because + consecutive PTOs might be caused by loss of packets or + acknowledgments due to severe congestion. Even when there are ack- + eliciting packets in flight in multiple packet number spaces, the + exponential increase in PTO occurs across all spaces to prevent + excess load on the network. For example, a timeout in the Initial + packet number space doubles the length of the timeout in the + Handshake packet number space. + + The total length of time over which consecutive PTOs expire is + limited by the idle timeout. + + The PTO timer MUST NOT be set if a timer is set for time threshold + loss detection; see Section 6.1.2. A timer that is set for time + threshold loss detection will expire earlier than the PTO timer in + most cases and is less likely to spuriously retransmit data. + +6.2.2. Handshakes and New Paths + + Resumed connections over the same network MAY use the previous + connection's final smoothed RTT value as the resumed connection's + initial RTT. When no previous RTT is available, the initial RTT + SHOULD be set to 333 milliseconds. This results in handshakes + starting with a PTO of 1 second, as recommended for TCP's initial + RTO; see Section 2 of [RFC6298]. + + A connection MAY use the delay between sending a PATH_CHALLENGE and + receiving a PATH_RESPONSE to set the initial RTT (see kInitialRtt in + Appendix A.2) for a new path, but the delay SHOULD NOT be considered + an RTT sample. + + When the Initial keys and Handshake keys are discarded (see + Section 6.4), any Initial packets and Handshake packets can no longer + be acknowledged, so they are removed from bytes in flight. When + Initial or Handshake keys are discarded, the PTO and loss detection + timers MUST be reset, because discarding keys indicates forward + progress and the loss detection timer might have been set for a now- + discarded packet number space. + +6.2.2.1. Before Address Validation + + Until the server has validated the client's address on the path, the + amount of data it can send is limited to three times the amount of + data received, as specified in Section 8.1 of [QUIC-TRANSPORT]. If + no additional data can be sent, the server's PTO timer MUST NOT be + armed until datagrams have been received from the client because + packets sent on PTO count against the anti-amplification limit. + + When the server receives a datagram from the client, the + amplification limit is increased and the server resets the PTO timer. + If the PTO timer is then set to a time in the past, it is executed + immediately. Doing so avoids sending new 1-RTT packets prior to + packets critical to the completion of the handshake. In particular, + this can happen when 0-RTT is accepted but the server fails to + validate the client's address. + + Since the server could be blocked until more datagrams are received + from the client, it is the client's responsibility to send packets to + unblock the server until it is certain that the server has finished + its address validation (see Section 8 of [QUIC-TRANSPORT]). That is, + the client MUST set the PTO timer if the client has not received an + acknowledgment for any of its Handshake packets and the handshake is + not confirmed (see Section 4.1.2 of [QUIC-TLS]), even if there are no + packets in flight. When the PTO fires, the client MUST send a + Handshake packet if it has Handshake keys, otherwise it MUST send an + Initial packet in a UDP datagram with a payload of at least 1200 + bytes. + +6.2.3. Speeding up Handshake Completion + + When a server receives an Initial packet containing duplicate CRYPTO + data, it can assume the client did not receive all of the server's + CRYPTO data sent in Initial packets, or the client's estimated RTT is + too small. When a client receives Handshake or 1-RTT packets prior + to obtaining Handshake keys, it may assume some or all of the + server's Initial packets were lost. + + To speed up handshake completion under these conditions, an endpoint + MAY, for a limited number of times per connection, send a packet + containing unacknowledged CRYPTO data earlier than the PTO expiry, + subject to the address validation limits in Section 8.1 of + [QUIC-TRANSPORT]. Doing so at most once for each connection is + adequate to quickly recover from a single packet loss. An endpoint + that always retransmits packets in response to receiving packets that + it cannot process risks creating an infinite exchange of packets. + + Endpoints can also use coalesced packets (see Section 12.2 of + [QUIC-TRANSPORT]) to ensure that each datagram elicits at least one + acknowledgment. For example, a client can coalesce an Initial packet + containing PING and PADDING frames with a 0-RTT data packet, and a + server can coalesce an Initial packet containing a PING frame with + one or more packets in its first flight. + +6.2.4. Sending Probe Packets + + When a PTO timer expires, a sender MUST send at least one ack- + eliciting packet in the packet number space as a probe. An endpoint + MAY send up to two full-sized datagrams containing ack-eliciting + packets to avoid an expensive consecutive PTO expiration due to a + single lost datagram or to transmit data from multiple packet number + spaces. All probe packets sent on a PTO MUST be ack-eliciting. + + In addition to sending data in the packet number space for which the + timer expired, the sender SHOULD send ack-eliciting packets from + other packet number spaces with in-flight data, coalescing packets if + possible. This is particularly valuable when the server has both + Initial and Handshake data in flight or when the client has both + Handshake and Application Data in flight because the peer might only + have receive keys for one of the two packet number spaces. + + If the sender wants to elicit a faster acknowledgment on PTO, it can + skip a packet number to eliminate the acknowledgment delay. + + An endpoint SHOULD include new data in packets that are sent on PTO + expiration. Previously sent data MAY be sent if no new data can be + sent. Implementations MAY use alternative strategies for determining + the content of probe packets, including sending new or retransmitted + data based on the application's priorities. + + It is possible the sender has no new or previously sent data to send. + As an example, consider the following sequence of events: new + application data is sent in a STREAM frame, deemed lost, then + retransmitted in a new packet, and then the original transmission is + acknowledged. When there is no data to send, the sender SHOULD send + a PING or other ack-eliciting frame in a single packet, rearming the + PTO timer. + + Alternatively, instead of sending an ack-eliciting packet, the sender + MAY mark any packets still in flight as lost. Doing so avoids + sending an additional packet but increases the risk that loss is + declared too aggressively, resulting in an unnecessary rate reduction + by the congestion controller. + + Consecutive PTO periods increase exponentially, and as a result, + connection recovery latency increases exponentially as packets + continue to be dropped in the network. Sending two packets on PTO + expiration increases resilience to packet drops, thus reducing the + probability of consecutive PTO events. + + When the PTO timer expires multiple times and new data cannot be + sent, implementations must choose between sending the same payload + every time or sending different payloads. Sending the same payload + may be simpler and ensures the highest priority frames arrive first. + Sending different payloads each time reduces the chances of spurious + retransmission. + +6.3. Handling Retry Packets + + A Retry packet causes a client to send another Initial packet, + effectively restarting the connection process. A Retry packet + indicates that the Initial packet was received but not processed. A + Retry packet cannot be treated as an acknowledgment because it does + not indicate that a packet was processed or specify the packet + number. + + Clients that receive a Retry packet reset congestion control and loss + recovery state, including resetting any pending timers. Other + connection state, in particular cryptographic handshake messages, is + retained; see Section 17.2.5 of [QUIC-TRANSPORT]. + + The client MAY compute an RTT estimate to the server as the time + period from when the first Initial packet was sent to when a Retry or + a Version Negotiation packet is received. The client MAY use this + value in place of its default for the initial RTT estimate. + +6.4. Discarding Keys and Packet State + + When Initial and Handshake packet protection keys are discarded (see + Section 4.9 of [QUIC-TLS]), all packets that were sent with those + keys can no longer be acknowledged because their acknowledgments + cannot be processed. The sender MUST discard all recovery state + associated with those packets and MUST remove them from the count of + bytes in flight. + + Endpoints stop sending and receiving Initial packets once they start + exchanging Handshake packets; see Section 17.2.2.1 of + [QUIC-TRANSPORT]. At this point, recovery state for all in-flight + Initial packets is discarded. + + When 0-RTT is rejected, recovery state for all in-flight 0-RTT + packets is discarded. + + If a server accepts 0-RTT, but does not buffer 0-RTT packets that + arrive before Initial packets, early 0-RTT packets will be declared + lost, but that is expected to be infrequent. + + It is expected that keys are discarded at some time after the packets + encrypted with them are either acknowledged or declared lost. + However, Initial and Handshake secrets are discarded as soon as + Handshake and 1-RTT keys are proven to be available to both client + and server; see Section 4.9.1 of [QUIC-TLS]. + +7. Congestion Control + + This document specifies a sender-side congestion controller for QUIC + similar to TCP NewReno [RFC6582]. + + The signals QUIC provides for congestion control are generic and are + designed to support different sender-side algorithms. A sender can + unilaterally choose a different algorithm to use, such as CUBIC + [RFC8312]. + + If a sender uses a different controller than that specified in this + document, the chosen controller MUST conform to the congestion + control guidelines specified in Section 3.1 of [RFC8085]. + + Similar to TCP, packets containing only ACK frames do not count + toward bytes in flight and are not congestion controlled. Unlike + TCP, QUIC can detect the loss of these packets and MAY use that + information to adjust the congestion controller or the rate of ACK- + only packets being sent, but this document does not describe a + mechanism for doing so. + + The congestion controller is per path, so packets sent on other paths + do not alter the current path's congestion controller, as described + in Section 9.4 of [QUIC-TRANSPORT]. + + The algorithm in this document specifies and uses the controller's + congestion window in bytes. + + An endpoint MUST NOT send a packet if it would cause bytes_in_flight + (see Appendix B.2) to be larger than the congestion window, unless + the packet is sent on a PTO timer expiration (see Section 6.2) or + when entering recovery (see Section 7.3.2). + +7.1. Explicit Congestion Notification + + If a path has been validated to support Explicit Congestion + Notification (ECN) [RFC3168] [RFC8311], QUIC treats a Congestion + Experienced (CE) codepoint in the IP header as a signal of + congestion. This document specifies an endpoint's response when the + peer-reported ECN-CE count increases; see Section 13.4.2 of + [QUIC-TRANSPORT]. + +7.2. Initial and Minimum Congestion Window + + QUIC begins every connection in slow start with the congestion window + set to an initial value. Endpoints SHOULD use an initial congestion + window of ten times the maximum datagram size (max_datagram_size), + while limiting the window to the larger of 14,720 bytes or twice the + maximum datagram size. This follows the analysis and recommendations + in [RFC6928], increasing the byte limit to account for the smaller + 8-byte overhead of UDP compared to the 20-byte overhead for TCP. + + If the maximum datagram size changes during the connection, the + initial congestion window SHOULD be recalculated with the new size. + If the maximum datagram size is decreased in order to complete the + handshake, the congestion window SHOULD be set to the new initial + congestion window. + + Prior to validating the client's address, the server can be further + limited by the anti-amplification limit as specified in Section 8.1 + of [QUIC-TRANSPORT]. Though the anti-amplification limit can prevent + the congestion window from being fully utilized and therefore slow + down the increase in congestion window, it does not directly affect + the congestion window. + + The minimum congestion window is the smallest value the congestion + window can attain in response to loss, an increase in the peer- + reported ECN-CE count, or persistent congestion. The RECOMMENDED + value is 2 * max_datagram_size. + +7.3. Congestion Control States + + The NewReno congestion controller described in this document has + three distinct states, as shown in Figure 1. + + New path or +------------+ + persistent congestion | Slow | + (O)---------------------->| Start | + +------------+ + | + Loss or | + ECN-CE increase | + v + +------------+ Loss or +------------+ + | Congestion | ECN-CE increase | Recovery | + | Avoidance |------------------>| Period | + +------------+ +------------+ + ^ | + | | + +----------------------------+ + Acknowledgment of packet + sent during recovery + + Figure 1: Congestion Control States and Transitions + + These states and the transitions between them are described in + subsequent sections. + +7.3.1. Slow Start + + A NewReno sender is in slow start any time the congestion window is + below the slow start threshold. A sender begins in slow start + because the slow start threshold is initialized to an infinite value. + + While a sender is in slow start, the congestion window increases by + the number of bytes acknowledged when each acknowledgment is + processed. This results in exponential growth of the congestion + window. + + The sender MUST exit slow start and enter a recovery period when a + packet is lost or when the ECN-CE count reported by its peer + increases. + + A sender reenters slow start any time the congestion window is less + than the slow start threshold, which only occurs after persistent + congestion is declared. + +7.3.2. Recovery + + A NewReno sender enters a recovery period when it detects the loss of + a packet or when the ECN-CE count reported by its peer increases. A + sender that is already in a recovery period stays in it and does not + reenter it. + + On entering a recovery period, a sender MUST set the slow start + threshold to half the value of the congestion window when loss is + detected. The congestion window MUST be set to the reduced value of + the slow start threshold before exiting the recovery period. + + Implementations MAY reduce the congestion window immediately upon + entering a recovery period or use other mechanisms, such as + Proportional Rate Reduction [PRR], to reduce the congestion window + more gradually. If the congestion window is reduced immediately, a + single packet can be sent prior to reduction. This speeds up loss + recovery if the data in the lost packet is retransmitted and is + similar to TCP as described in Section 5 of [RFC6675]. + + The recovery period aims to limit congestion window reduction to once + per round trip. Therefore, during a recovery period, the congestion + window does not change in response to new losses or increases in the + ECN-CE count. + + A recovery period ends and the sender enters congestion avoidance + when a packet sent during the recovery period is acknowledged. This + is slightly different from TCP's definition of recovery, which ends + when the lost segment that started recovery is acknowledged + [RFC5681]. + +7.3.3. Congestion Avoidance + + A NewReno sender is in congestion avoidance any time the congestion + window is at or above the slow start threshold and not in a recovery + period. + + A sender in congestion avoidance uses an Additive Increase + Multiplicative Decrease (AIMD) approach that MUST limit the increase + to the congestion window to at most one maximum datagram size for + each congestion window that is acknowledged. + + The sender exits congestion avoidance and enters a recovery period + when a packet is lost or when the ECN-CE count reported by its peer + increases. + +7.4. Ignoring Loss of Undecryptable Packets + + During the handshake, some packet protection keys might not be + available when a packet arrives, and the receiver can choose to drop + the packet. In particular, Handshake and 0-RTT packets cannot be + processed until the Initial packets arrive, and 1-RTT packets cannot + be processed until the handshake completes. Endpoints MAY ignore the + loss of Handshake, 0-RTT, and 1-RTT packets that might have arrived + before the peer had packet protection keys to process those packets. + Endpoints MUST NOT ignore the loss of packets that were sent after + the earliest acknowledged packet in a given packet number space. + +7.5. Probe Timeout + + Probe packets MUST NOT be blocked by the congestion controller. A + sender MUST however count these packets as being additionally in + flight, since these packets add network load without establishing + packet loss. Note that sending probe packets might cause the + sender's bytes in flight to exceed the congestion window until an + acknowledgment is received that establishes loss or delivery of + packets. + +7.6. Persistent Congestion + + When a sender establishes loss of all packets sent over a long enough + duration, the network is considered to be experiencing persistent + congestion. + +7.6.1. Duration + + The persistent congestion duration is computed as follows: + + (smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay) * + kPersistentCongestionThreshold + + Unlike the PTO computation in Section 6.2, this duration includes the + max_ack_delay irrespective of the packet number spaces in which + losses are established. + + This duration allows a sender to send as many packets before + establishing persistent congestion, including some in response to PTO + expiration, as TCP does with Tail Loss Probes [RFC8985] and an RTO + [RFC5681]. + + Larger values of kPersistentCongestionThreshold cause the sender to + become less responsive to persistent congestion in the network, which + can result in aggressive sending into a congested network. Too small + a value can result in a sender declaring persistent congestion + unnecessarily, resulting in reduced throughput for the sender. + + The RECOMMENDED value for kPersistentCongestionThreshold is 3, which + results in behavior that is approximately equivalent to a TCP sender + declaring an RTO after two TLPs. + + This design does not use consecutive PTO events to establish + persistent congestion, since application patterns impact PTO + expiration. For example, a sender that sends small amounts of data + with silence periods between them restarts the PTO timer every time + it sends, potentially preventing the PTO timer from expiring for a + long period of time, even when no acknowledgments are being received. + The use of a duration enables a sender to establish persistent + congestion without depending on PTO expiration. + +7.6.2. Establishing Persistent Congestion + + A sender establishes persistent congestion after the receipt of an + acknowledgment if two packets that are ack-eliciting are declared + lost, and: + + * across all packet number spaces, none of the packets sent between + the send times of these two packets are acknowledged; + + * the duration between the send times of these two packets exceeds + the persistent congestion duration (Section 7.6.1); and + + * a prior RTT sample existed when these two packets were sent. + + These two packets MUST be ack-eliciting, since a receiver is required + to acknowledge only ack-eliciting packets within its maximum + acknowledgment delay; see Section 13.2 of [QUIC-TRANSPORT]. + + The persistent congestion period SHOULD NOT start until there is at + least one RTT sample. Before the first RTT sample, a sender arms its + PTO timer based on the initial RTT (Section 6.2.2), which could be + substantially larger than the actual RTT. Requiring a prior RTT + sample prevents a sender from establishing persistent congestion with + potentially too few probes. + + Since network congestion is not affected by packet number spaces, + persistent congestion SHOULD consider packets sent across packet + number spaces. A sender that does not have state for all packet + number spaces or an implementation that cannot compare send times + across packet number spaces MAY use state for just the packet number + space that was acknowledged. This might result in erroneously + declaring persistent congestion, but it will not lead to a failure to + detect persistent congestion. + + When persistent congestion is declared, the sender's congestion + window MUST be reduced to the minimum congestion window + (kMinimumWindow), similar to a TCP sender's response on an RTO + [RFC5681]. + +7.6.3. Example + + The following example illustrates how a sender might establish + persistent congestion. Assume: + + smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay = 2 + kPersistentCongestionThreshold = 3 + + Consider the following sequence of events: + + +========+===================================+ + | Time | Action | + +========+===================================+ + | t=0 | Send packet #1 (application data) | + +--------+-----------------------------------+ + | t=1 | Send packet #2 (application data) | + +--------+-----------------------------------+ + | t=1.2 | Receive acknowledgment of #1 | + +--------+-----------------------------------+ + | t=2 | Send packet #3 (application data) | + +--------+-----------------------------------+ + | t=3 | Send packet #4 (application data) | + +--------+-----------------------------------+ + | t=4 | Send packet #5 (application data) | + +--------+-----------------------------------+ + | t=5 | Send packet #6 (application data) | + +--------+-----------------------------------+ + | t=6 | Send packet #7 (application data) | + +--------+-----------------------------------+ + | t=8 | Send packet #8 (PTO 1) | + +--------+-----------------------------------+ + | t=12 | Send packet #9 (PTO 2) | + +--------+-----------------------------------+ + | t=12.2 | Receive acknowledgment of #9 | + +--------+-----------------------------------+ + + Table 1 + + Packets 2 through 8 are declared lost when the acknowledgment for + packet 9 is received at "t = 12.2". + + The congestion period is calculated as the time between the oldest + and newest lost packets: "8 - 1 = 7". The persistent congestion + duration is "2 * 3 = 6". Because the threshold was reached and + because none of the packets between the oldest and the newest lost + packets were acknowledged, the network is considered to have + experienced persistent congestion. + + While this example shows PTO expiration, they are not required for + persistent congestion to be established. + +7.7. Pacing + + A sender SHOULD pace sending of all in-flight packets based on input + from the congestion controller. + + Sending multiple packets into the network without any delay between + them creates a packet burst that might cause short-term congestion + and losses. Senders MUST either use pacing or limit such bursts. + Senders SHOULD limit bursts to the initial congestion window; see + Section 7.2. A sender with knowledge that the network path to the + receiver can absorb larger bursts MAY use a higher limit. + + An implementation should take care to architect its congestion + controller to work well with a pacer. For instance, a pacer might + wrap the congestion controller and control the availability of the + congestion window, or a pacer might pace out packets handed to it by + the congestion controller. + + Timely delivery of ACK frames is important for efficient loss + recovery. To avoid delaying their delivery to the peer, packets + containing only ACK frames SHOULD therefore not be paced. + + Endpoints can implement pacing as they choose. A perfectly paced + sender spreads packets exactly evenly over time. For a window-based + congestion controller, such as the one in this document, that rate + can be computed by averaging the congestion window over the RTT. + Expressed as a rate in units of bytes per time, where + congestion_window is in bytes: + + rate = N * congestion_window / smoothed_rtt + + Or expressed as an inter-packet interval in units of time: + + interval = ( smoothed_rtt * packet_size / congestion_window ) / N + + Using a value for "N" that is small, but at least 1 (for example, + 1.25) ensures that variations in RTT do not result in + underutilization of the congestion window. + + Practical considerations, such as packetization, scheduling delays, + and computational efficiency, can cause a sender to deviate from this + rate over time periods that are much shorter than an RTT. + + One possible implementation strategy for pacing uses a leaky bucket + algorithm, where the capacity of the "bucket" is limited to the + maximum burst size and the rate the "bucket" fills is determined by + the above function. + +7.8. Underutilizing the Congestion Window + + When bytes in flight is smaller than the congestion window and + sending is not pacing limited, the congestion window is + underutilized. This can happen due to insufficient application data + or flow control limits. When this occurs, the congestion window + SHOULD NOT be increased in either slow start or congestion avoidance. + + A sender that paces packets (see Section 7.7) might delay sending + packets and not fully utilize the congestion window due to this + delay. A sender SHOULD NOT consider itself application limited if it + would have fully utilized the congestion window without pacing delay. + + A sender MAY implement alternative mechanisms to update its + congestion window after periods of underutilization, such as those + proposed for TCP in [RFC7661]. + +8. Security Considerations + +8.1. Loss and Congestion Signals + + Loss detection and congestion control fundamentally involve the + consumption of signals, such as delay, loss, and ECN markings, from + unauthenticated entities. An attacker can cause endpoints to reduce + their sending rate by manipulating these signals: by dropping + packets, by altering path delay strategically, or by changing ECN + codepoints. + +8.2. Traffic Analysis + + Packets that carry only ACK frames can be heuristically identified by + observing packet size. Acknowledgment patterns may expose + information about link characteristics or application behavior. To + reduce leaked information, endpoints can bundle acknowledgments with + other frames, or they can use PADDING frames at a potential cost to + performance. + +8.3. Misreporting ECN Markings + + A receiver can misreport ECN markings to alter the congestion + response of a sender. Suppressing reports of ECN-CE markings could + cause a sender to increase their send rate. This increase could + result in congestion and loss. + + A sender can detect suppression of reports by marking occasional + packets that it sends with an ECN-CE marking. If a packet sent with + an ECN-CE marking is not reported as having been CE marked when the + packet is acknowledged, then the sender can disable ECN for that path + by not setting ECN-Capable Transport (ECT) codepoints in subsequent + packets sent on that path [RFC3168]. + + Reporting additional ECN-CE markings will cause a sender to reduce + their sending rate, which is similar in effect to advertising reduced + connection flow control limits and so no advantage is gained by doing + so. + + Endpoints choose the congestion controller that they use. Congestion + controllers respond to reports of ECN-CE by reducing their rate, but + the response may vary. Markings can be treated as equivalent to loss + [RFC3168], but other responses can be specified, such as [RFC8511] or + [RFC8311]. + +9. References + +9.1. Normative References + + [QUIC-TLS] Thomson, M., Ed. and S. Turner, Ed., "Using TLS to Secure + QUIC", RFC 9001, DOI 10.17487/RFC9001, May 2021, + <https://www.rfc-editor.org/info/rfc9001>. + + [QUIC-TRANSPORT] + Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC3168] Ramakrishnan, K., Floyd, S., and D. Black, "The Addition + of Explicit Congestion Notification (ECN) to IP", + RFC 3168, DOI 10.17487/RFC3168, September 2001, + <https://www.rfc-editor.org/info/rfc3168>. + + [RFC8085] Eggert, L., Fairhurst, G., and G. Shepherd, "UDP Usage + Guidelines", BCP 145, RFC 8085, DOI 10.17487/RFC8085, + March 2017, <https://www.rfc-editor.org/info/rfc8085>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + +9.2. Informative References + + [FACK] Mathis, M. and J. Mahdavi, "Forward acknowledgement: + Refining TCP Congestion Control", ACM SIGCOMM Computer + Communication Review, DOI 10.1145/248157.248181, August + 1996, <https://doi.org/10.1145/248157.248181>. + + [PRR] Mathis, M., Dukkipati, N., and Y. Cheng, "Proportional + Rate Reduction for TCP", RFC 6937, DOI 10.17487/RFC6937, + May 2013, <https://www.rfc-editor.org/info/rfc6937>. + + [RETRANSMISSION] + Karn, P. and C. Partridge, "Improving Round-Trip Time + Estimates in Reliable Transport Protocols", ACM + Transactions on Computer Systems, + DOI 10.1145/118544.118549, November 1991, + <https://doi.org/10.1145/118544.118549>. + + [RFC2018] Mathis, M., Mahdavi, J., Floyd, S., and A. Romanow, "TCP + Selective Acknowledgment Options", RFC 2018, + DOI 10.17487/RFC2018, October 1996, + <https://www.rfc-editor.org/info/rfc2018>. + + [RFC3465] Allman, M., "TCP Congestion Control with Appropriate Byte + Counting (ABC)", RFC 3465, DOI 10.17487/RFC3465, February + 2003, <https://www.rfc-editor.org/info/rfc3465>. + + [RFC5681] Allman, M., Paxson, V., and E. Blanton, "TCP Congestion + Control", RFC 5681, DOI 10.17487/RFC5681, September 2009, + <https://www.rfc-editor.org/info/rfc5681>. + + [RFC5682] Sarolahti, P., Kojo, M., Yamamoto, K., and M. Hata, + "Forward RTO-Recovery (F-RTO): An Algorithm for Detecting + Spurious Retransmission Timeouts with TCP", RFC 5682, + DOI 10.17487/RFC5682, September 2009, + <https://www.rfc-editor.org/info/rfc5682>. + + [RFC5827] Allman, M., Avrachenkov, K., Ayesta, U., Blanton, J., and + P. Hurtig, "Early Retransmit for TCP and Stream Control + Transmission Protocol (SCTP)", RFC 5827, + DOI 10.17487/RFC5827, May 2010, + <https://www.rfc-editor.org/info/rfc5827>. + + [RFC6298] Paxson, V., Allman, M., Chu, J., and M. Sargent, + "Computing TCP's Retransmission Timer", RFC 6298, + DOI 10.17487/RFC6298, June 2011, + <https://www.rfc-editor.org/info/rfc6298>. + + [RFC6582] Henderson, T., Floyd, S., Gurtov, A., and Y. Nishida, "The + NewReno Modification to TCP's Fast Recovery Algorithm", + RFC 6582, DOI 10.17487/RFC6582, April 2012, + <https://www.rfc-editor.org/info/rfc6582>. + + [RFC6675] Blanton, E., Allman, M., Wang, L., Jarvinen, I., Kojo, M., + and Y. Nishida, "A Conservative Loss Recovery Algorithm + Based on Selective Acknowledgment (SACK) for TCP", + RFC 6675, DOI 10.17487/RFC6675, August 2012, + <https://www.rfc-editor.org/info/rfc6675>. + + [RFC6928] Chu, J., Dukkipati, N., Cheng, Y., and M. Mathis, + "Increasing TCP's Initial Window", RFC 6928, + DOI 10.17487/RFC6928, April 2013, + <https://www.rfc-editor.org/info/rfc6928>. + + [RFC7661] Fairhurst, G., Sathiaseelan, A., and R. Secchi, "Updating + TCP to Support Rate-Limited Traffic", RFC 7661, + DOI 10.17487/RFC7661, October 2015, + <https://www.rfc-editor.org/info/rfc7661>. + + [RFC8311] Black, D., "Relaxing Restrictions on Explicit Congestion + Notification (ECN) Experimentation", RFC 8311, + DOI 10.17487/RFC8311, January 2018, + <https://www.rfc-editor.org/info/rfc8311>. + + [RFC8312] Rhee, I., Xu, L., Ha, S., Zimmermann, A., Eggert, L., and + R. Scheffenegger, "CUBIC for Fast Long-Distance Networks", + RFC 8312, DOI 10.17487/RFC8312, February 2018, + <https://www.rfc-editor.org/info/rfc8312>. + + [RFC8511] Khademi, N., Welzl, M., Armitage, G., and G. Fairhurst, + "TCP Alternative Backoff with ECN (ABE)", RFC 8511, + DOI 10.17487/RFC8511, December 2018, + <https://www.rfc-editor.org/info/rfc8511>. + + [RFC8985] Cheng, Y., Cardwell, N., Dukkipati, N., and P. Jha, "The + RACK-TLP Loss Detection Algorithm for TCP", RFC 8985, + DOI 10.17487/RFC8985, February 2021, + <https://www.rfc-editor.org/info/rfc8985>. + +Appendix A. Loss Recovery Pseudocode + + We now describe an example implementation of the loss detection + mechanisms described in Section 6. + + The pseudocode segments in this section are licensed as Code + Components; see the copyright notice. + +A.1. Tracking Sent Packets + + To correctly implement congestion control, a QUIC sender tracks every + ack-eliciting packet until the packet is acknowledged or lost. It is + expected that implementations will be able to access this information + by packet number and crypto context and store the per-packet fields + (Appendix A.1.1) for loss recovery and congestion control. + + After a packet is declared lost, the endpoint can still maintain + state for it for an amount of time to allow for packet reordering; + see Section 13.3 of [QUIC-TRANSPORT]. This enables a sender to + detect spurious retransmissions. + + Sent packets are tracked for each packet number space, and ACK + processing only applies to a single space. + +A.1.1. Sent Packet Fields + + packet_number: The packet number of the sent packet. + + ack_eliciting: A Boolean that indicates whether a packet is ack- + eliciting. If true, it is expected that an acknowledgment will be + received, though the peer could delay sending the ACK frame + containing it by up to the max_ack_delay. + + in_flight: A Boolean that indicates whether the packet counts toward + bytes in flight. + + sent_bytes: The number of bytes sent in the packet, not including + UDP or IP overhead, but including QUIC framing overhead. + + time_sent: The time the packet was sent. + +A.2. Constants of Interest + + Constants used in loss recovery are based on a combination of RFCs, + papers, and common practice. + + kPacketThreshold: Maximum reordering in packets before packet + threshold loss detection considers a packet lost. The value + recommended in Section 6.1.1 is 3. + + kTimeThreshold: Maximum reordering in time before time threshold + loss detection considers a packet lost. Specified as an RTT + multiplier. The value recommended in Section 6.1.2 is 9/8. + + kGranularity: Timer granularity. This is a system-dependent value, + and Section 6.1.2 recommends a value of 1 ms. + + kInitialRtt: The RTT used before an RTT sample is taken. The value + recommended in Section 6.2.2 is 333 ms. + + kPacketNumberSpace: An enum to enumerate the three packet number + spaces: + + enum kPacketNumberSpace { + Initial, + Handshake, + ApplicationData, + } + +A.3. Variables of Interest + + Variables required to implement the congestion control mechanisms are + described in this section. + + latest_rtt: The most recent RTT measurement made when receiving an + acknowledgment for a previously unacknowledged packet. + + smoothed_rtt: The smoothed RTT of the connection, computed as + described in Section 5.3. + + rttvar: The RTT variation, computed as described in Section 5.3. + + min_rtt: The minimum RTT seen over a period of time, ignoring + acknowledgment delay, as described in Section 5.2. + + first_rtt_sample: The time that the first RTT sample was obtained. + + max_ack_delay: The maximum amount of time by which the receiver + intends to delay acknowledgments for packets in the Application + Data packet number space, as defined by the eponymous transport + parameter (Section 18.2 of [QUIC-TRANSPORT]). Note that the + actual ack_delay in a received ACK frame may be larger due to late + timers, reordering, or loss. + + loss_detection_timer: Multi-modal timer used for loss detection. + + pto_count: The number of times a PTO has been sent without receiving + an acknowledgment. + + time_of_last_ack_eliciting_packet[kPacketNumberSpace]: The time the + most recent ack-eliciting packet was sent. + + largest_acked_packet[kPacketNumberSpace]: The largest packet number + acknowledged in the packet number space so far. + + loss_time[kPacketNumberSpace]: The time at which the next packet in + that packet number space can be considered lost based on exceeding + the reordering window in time. + + sent_packets[kPacketNumberSpace]: An association of packet numbers + in a packet number space to information about them. Described in + detail above in Appendix A.1. + +A.4. Initialization + + At the beginning of the connection, initialize the loss detection + variables as follows: + + loss_detection_timer.reset() + pto_count = 0 + latest_rtt = 0 + smoothed_rtt = kInitialRtt + rttvar = kInitialRtt / 2 + min_rtt = 0 + first_rtt_sample = 0 + for pn_space in [ Initial, Handshake, ApplicationData ]: + largest_acked_packet[pn_space] = infinite + time_of_last_ack_eliciting_packet[pn_space] = 0 + loss_time[pn_space] = 0 + +A.5. On Sending a Packet + + After a packet is sent, information about the packet is stored. The + parameters to OnPacketSent are described in detail above in + Appendix A.1.1. + + Pseudocode for OnPacketSent follows: + + OnPacketSent(packet_number, pn_space, ack_eliciting, + in_flight, sent_bytes): + sent_packets[pn_space][packet_number].packet_number = + packet_number + sent_packets[pn_space][packet_number].time_sent = now() + sent_packets[pn_space][packet_number].ack_eliciting = + ack_eliciting + sent_packets[pn_space][packet_number].in_flight = in_flight + sent_packets[pn_space][packet_number].sent_bytes = sent_bytes + if (in_flight): + if (ack_eliciting): + time_of_last_ack_eliciting_packet[pn_space] = now() + OnPacketSentCC(sent_bytes) + SetLossDetectionTimer() + +A.6. On Receiving a Datagram + + When a server is blocked by anti-amplification limits, receiving a + datagram unblocks it, even if none of the packets in the datagram are + successfully processed. In such a case, the PTO timer will need to + be rearmed. + + Pseudocode for OnDatagramReceived follows: + + OnDatagramReceived(datagram): + // If this datagram unblocks the server, arm the + // PTO timer to avoid deadlock. + if (server was at anti-amplification limit): + SetLossDetectionTimer() + if loss_detection_timer.timeout < now(): + // Execute PTO if it would have expired + // while the amplification limit applied. + OnLossDetectionTimeout() + +A.7. On Receiving an Acknowledgment + + When an ACK frame is received, it may newly acknowledge any number of + packets. + + Pseudocode for OnAckReceived and UpdateRtt follow: + + IncludesAckEliciting(packets): + for packet in packets: + if (packet.ack_eliciting): + return true + return false + + OnAckReceived(ack, pn_space): + if (largest_acked_packet[pn_space] == infinite): + largest_acked_packet[pn_space] = ack.largest_acked + else: + largest_acked_packet[pn_space] = + max(largest_acked_packet[pn_space], ack.largest_acked) + + // DetectAndRemoveAckedPackets finds packets that are newly + // acknowledged and removes them from sent_packets. + newly_acked_packets = + DetectAndRemoveAckedPackets(ack, pn_space) + // Nothing to do if there are no newly acked packets. + if (newly_acked_packets.empty()): + return + + // Update the RTT if the largest acknowledged is newly acked + // and at least one ack-eliciting was newly acked. + if (newly_acked_packets.largest().packet_number == + ack.largest_acked && + IncludesAckEliciting(newly_acked_packets)): + latest_rtt = + now() - newly_acked_packets.largest().time_sent + UpdateRtt(ack.ack_delay) + + // Process ECN information if present. + if (ACK frame contains ECN information): + ProcessECN(ack, pn_space) + + lost_packets = DetectAndRemoveLostPackets(pn_space) + if (!lost_packets.empty()): + OnPacketsLost(lost_packets) + OnPacketsAcked(newly_acked_packets) + + // Reset pto_count unless the client is unsure if + // the server has validated the client's address. + if (PeerCompletedAddressValidation()): + pto_count = 0 + SetLossDetectionTimer() + + + UpdateRtt(ack_delay): + if (first_rtt_sample == 0): + min_rtt = latest_rtt + smoothed_rtt = latest_rtt + rttvar = latest_rtt / 2 + first_rtt_sample = now() + return + + // min_rtt ignores acknowledgment delay. + min_rtt = min(min_rtt, latest_rtt) + // Limit ack_delay by max_ack_delay after handshake + // confirmation. + if (handshake confirmed): + ack_delay = min(ack_delay, max_ack_delay) + + // Adjust for acknowledgment delay if plausible. + adjusted_rtt = latest_rtt + if (latest_rtt >= min_rtt + ack_delay): + adjusted_rtt = latest_rtt - ack_delay + + rttvar = 3/4 * rttvar + 1/4 * abs(smoothed_rtt - adjusted_rtt) + smoothed_rtt = 7/8 * smoothed_rtt + 1/8 * adjusted_rtt + +A.8. Setting the Loss Detection Timer + + QUIC loss detection uses a single timer for all timeout loss + detection. The duration of the timer is based on the timer's mode, + which is set in the packet and timer events further below. The + function SetLossDetectionTimer defined below shows how the single + timer is set. + + This algorithm may result in the timer being set in the past, + particularly if timers wake up late. Timers set in the past fire + immediately. + + Pseudocode for SetLossDetectionTimer follows (where the "^" operator + represents exponentiation): + + GetLossTimeAndSpace(): + time = loss_time[Initial] + space = Initial + for pn_space in [ Handshake, ApplicationData ]: + if (time == 0 || loss_time[pn_space] < time): + time = loss_time[pn_space]; + space = pn_space + return time, space + + GetPtoTimeAndSpace(): + duration = (smoothed_rtt + max(4 * rttvar, kGranularity)) + * (2 ^ pto_count) + // Anti-deadlock PTO starts from the current time + if (no ack-eliciting packets in flight): + assert(!PeerCompletedAddressValidation()) + if (has handshake keys): + return (now() + duration), Handshake + else: + return (now() + duration), Initial + pto_timeout = infinite + pto_space = Initial + for space in [ Initial, Handshake, ApplicationData ]: + if (no ack-eliciting packets in flight in space): + continue; + if (space == ApplicationData): + // Skip Application Data until handshake confirmed. + if (handshake is not confirmed): + return pto_timeout, pto_space + // Include max_ack_delay and backoff for Application Data. + duration += max_ack_delay * (2 ^ pto_count) + + t = time_of_last_ack_eliciting_packet[space] + duration + if (t < pto_timeout): + pto_timeout = t + pto_space = space + return pto_timeout, pto_space + + PeerCompletedAddressValidation(): + // Assume clients validate the server's address implicitly. + if (endpoint is server): + return true + // Servers complete address validation when a + // protected packet is received. + return has received Handshake ACK || + handshake confirmed + + SetLossDetectionTimer(): + earliest_loss_time, _ = GetLossTimeAndSpace() + if (earliest_loss_time != 0): + // Time threshold loss detection. + loss_detection_timer.update(earliest_loss_time) + return + + if (server is at anti-amplification limit): + // The server's timer is not set if nothing can be sent. + loss_detection_timer.cancel() + return + + if (no ack-eliciting packets in flight && + PeerCompletedAddressValidation()): + // There is nothing to detect lost, so no timer is set. + // However, the client needs to arm the timer if the + // server might be blocked by the anti-amplification limit. + loss_detection_timer.cancel() + return + + timeout, _ = GetPtoTimeAndSpace() + loss_detection_timer.update(timeout) + +A.9. On Timeout + + When the loss detection timer expires, the timer's mode determines + the action to be performed. + + Pseudocode for OnLossDetectionTimeout follows: + + OnLossDetectionTimeout(): + earliest_loss_time, pn_space = GetLossTimeAndSpace() + if (earliest_loss_time != 0): + // Time threshold loss Detection + lost_packets = DetectAndRemoveLostPackets(pn_space) + assert(!lost_packets.empty()) + OnPacketsLost(lost_packets) + SetLossDetectionTimer() + return + + if (no ack-eliciting packets in flight): + assert(!PeerCompletedAddressValidation()) + // Client sends an anti-deadlock packet: Initial is padded + // to earn more anti-amplification credit, + // a Handshake packet proves address ownership. + if (has Handshake keys): + SendOneAckElicitingHandshakePacket() + else: + SendOneAckElicitingPaddedInitialPacket() + else: + // PTO. Send new data if available, else retransmit old data. + // If neither is available, send a single PING frame. + _, pn_space = GetPtoTimeAndSpace() + SendOneOrTwoAckElicitingPackets(pn_space) + + pto_count++ + SetLossDetectionTimer() + +A.10. Detecting Lost Packets + + DetectAndRemoveLostPackets is called every time an ACK is received or + the time threshold loss detection timer expires. This function + operates on the sent_packets for that packet number space and returns + a list of packets newly detected as lost. + + Pseudocode for DetectAndRemoveLostPackets follows: + + DetectAndRemoveLostPackets(pn_space): + assert(largest_acked_packet[pn_space] != infinite) + loss_time[pn_space] = 0 + lost_packets = [] + loss_delay = kTimeThreshold * max(latest_rtt, smoothed_rtt) + + // Minimum time of kGranularity before packets are deemed lost. + loss_delay = max(loss_delay, kGranularity) + + // Packets sent before this time are deemed lost. + lost_send_time = now() - loss_delay + + foreach unacked in sent_packets[pn_space]: + if (unacked.packet_number > largest_acked_packet[pn_space]): + continue + + // Mark packet as lost, or set time when it should be marked. + // Note: The use of kPacketThreshold here assumes that there + // were no sender-induced gaps in the packet number space. + if (unacked.time_sent <= lost_send_time || + largest_acked_packet[pn_space] >= + unacked.packet_number + kPacketThreshold): + sent_packets[pn_space].remove(unacked.packet_number) + lost_packets.insert(unacked) + else: + if (loss_time[pn_space] == 0): + loss_time[pn_space] = unacked.time_sent + loss_delay + else: + loss_time[pn_space] = min(loss_time[pn_space], + unacked.time_sent + loss_delay) + return lost_packets + +A.11. Upon Dropping Initial or Handshake Keys + + When Initial or Handshake keys are discarded, packets from the space + are discarded and loss detection state is updated. + + Pseudocode for OnPacketNumberSpaceDiscarded follows: + + OnPacketNumberSpaceDiscarded(pn_space): + assert(pn_space != ApplicationData) + RemoveFromBytesInFlight(sent_packets[pn_space]) + sent_packets[pn_space].clear() + // Reset the loss detection and PTO timer + time_of_last_ack_eliciting_packet[pn_space] = 0 + loss_time[pn_space] = 0 + pto_count = 0 + SetLossDetectionTimer() + +Appendix B. Congestion Control Pseudocode + + We now describe an example implementation of the congestion + controller described in Section 7. + + The pseudocode segments in this section are licensed as Code + Components; see the copyright notice. + +B.1. Constants of Interest + + Constants used in congestion control are based on a combination of + RFCs, papers, and common practice. + + kInitialWindow: Default limit on the initial bytes in flight as + described in Section 7.2. + + kMinimumWindow: Minimum congestion window in bytes as described in + Section 7.2. + + kLossReductionFactor: Scaling factor applied to reduce the + congestion window when a new loss event is detected. Section 7 + recommends a value of 0.5. + + kPersistentCongestionThreshold: Period of time for persistent + congestion to be established, specified as a PTO multiplier. + Section 7.6 recommends a value of 3. + +B.2. Variables of Interest + + Variables required to implement the congestion control mechanisms are + described in this section. + + max_datagram_size: The sender's current maximum payload size. This + does not include UDP or IP overhead. The max datagram size is + used for congestion window computations. An endpoint sets the + value of this variable based on its Path Maximum Transmission Unit + (PMTU; see Section 14.2 of [QUIC-TRANSPORT]), with a minimum value + of 1200 bytes. + + ecn_ce_counters[kPacketNumberSpace]: The highest value reported for + the ECN-CE counter in the packet number space by the peer in an + ACK frame. This value is used to detect increases in the reported + ECN-CE counter. + + bytes_in_flight: The sum of the size in bytes of all sent packets + that contain at least one ack-eliciting or PADDING frame and have + not been acknowledged or declared lost. The size does not include + IP or UDP overhead, but does include the QUIC header and + Authenticated Encryption with Associated Data (AEAD) overhead. + Packets only containing ACK frames do not count toward + bytes_in_flight to ensure congestion control does not impede + congestion feedback. + + congestion_window: Maximum number of bytes allowed to be in flight. + + congestion_recovery_start_time: The time the current recovery period + started due to the detection of loss or ECN. When a packet sent + after this time is acknowledged, QUIC exits congestion recovery. + + ssthresh: Slow start threshold in bytes. When the congestion window + is below ssthresh, the mode is slow start and the window grows by + the number of bytes acknowledged. + + The congestion control pseudocode also accesses some of the variables + from the loss recovery pseudocode. + +B.3. Initialization + + At the beginning of the connection, initialize the congestion control + variables as follows: + + congestion_window = kInitialWindow + bytes_in_flight = 0 + congestion_recovery_start_time = 0 + ssthresh = infinite + for pn_space in [ Initial, Handshake, ApplicationData ]: + ecn_ce_counters[pn_space] = 0 + +B.4. On Packet Sent + + Whenever a packet is sent and it contains non-ACK frames, the packet + increases bytes_in_flight. + + OnPacketSentCC(sent_bytes): + bytes_in_flight += sent_bytes + +B.5. On Packet Acknowledgment + + This is invoked from loss detection's OnAckReceived and is supplied + with the newly acked_packets from sent_packets. + + In congestion avoidance, implementers that use an integer + representation for congestion_window should be careful with division + and can use the alternative approach suggested in Section 2.1 of + [RFC3465]. + + InCongestionRecovery(sent_time): + return sent_time <= congestion_recovery_start_time + + OnPacketsAcked(acked_packets): + for acked_packet in acked_packets: + OnPacketAcked(acked_packet) + + OnPacketAcked(acked_packet): + if (!acked_packet.in_flight): + return; + // Remove from bytes_in_flight. + bytes_in_flight -= acked_packet.sent_bytes + // Do not increase congestion_window if application + // limited or flow control limited. + if (IsAppOrFlowControlLimited()) + return + // Do not increase congestion window in recovery period. + if (InCongestionRecovery(acked_packet.time_sent)): + return + if (congestion_window < ssthresh): + // Slow start. + congestion_window += acked_packet.sent_bytes + else: + // Congestion avoidance. + congestion_window += + max_datagram_size * acked_packet.sent_bytes + / congestion_window + +B.6. On New Congestion Event + + This is invoked from ProcessECN and OnPacketsLost when a new + congestion event is detected. If not already in recovery, this + starts a recovery period and reduces the slow start threshold and + congestion window immediately. + + OnCongestionEvent(sent_time): + // No reaction if already in a recovery period. + if (InCongestionRecovery(sent_time)): + return + + // Enter recovery period. + congestion_recovery_start_time = now() + ssthresh = congestion_window * kLossReductionFactor + congestion_window = max(ssthresh, kMinimumWindow) + // A packet can be sent to speed up loss recovery. + MaybeSendOnePacket() + +B.7. Process ECN Information + + This is invoked when an ACK frame with an ECN section is received + from the peer. + + ProcessECN(ack, pn_space): + // If the ECN-CE counter reported by the peer has increased, + // this could be a new congestion event. + if (ack.ce_counter > ecn_ce_counters[pn_space]): + ecn_ce_counters[pn_space] = ack.ce_counter + sent_time = sent_packets[ack.largest_acked].time_sent + OnCongestionEvent(sent_time) + +B.8. On Packets Lost + + This is invoked when DetectAndRemoveLostPackets deems packets lost. + + OnPacketsLost(lost_packets): + sent_time_of_last_loss = 0 + // Remove lost packets from bytes_in_flight. + for lost_packet in lost_packets: + if lost_packet.in_flight: + bytes_in_flight -= lost_packet.sent_bytes + sent_time_of_last_loss = + max(sent_time_of_last_loss, lost_packet.time_sent) + // Congestion event if in-flight packets were lost + if (sent_time_of_last_loss != 0): + OnCongestionEvent(sent_time_of_last_loss) + + // Reset the congestion window if the loss of these + // packets indicates persistent congestion. + // Only consider packets sent after getting an RTT sample. + if (first_rtt_sample == 0): + return + pc_lost = [] + for lost in lost_packets: + if lost.time_sent > first_rtt_sample: + pc_lost.insert(lost) + if (InPersistentCongestion(pc_lost)): + congestion_window = kMinimumWindow + congestion_recovery_start_time = 0 + +B.9. Removing Discarded Packets from Bytes in Flight + + When Initial or Handshake keys are discarded, packets sent in that + space no longer count toward bytes in flight. + + Pseudocode for RemoveFromBytesInFlight follows: + + RemoveFromBytesInFlight(discarded_packets): + // Remove any unacknowledged packets from flight. + foreach packet in discarded_packets: + if packet.in_flight + bytes_in_flight -= size + +Contributors + + The IETF QUIC Working Group received an enormous amount of support + from many people. The following people provided substantive + contributions to this document: + + * Alessandro Ghedini + * Benjamin Saunders + * Gorry Fairhurst + * ๅฑฑๆœฌๅ’Œๅฝฆ (Kazu Yamamoto) + * ๅฅฅ ไธ€็ฉ‚ (Kazuho Oku) + * Lars Eggert + * Magnus Westerlund + * Marten Seemann + * Martin Duke + * Martin Thomson + * Mirja Kรผhlewind + * Nick Banks + * Praveen Balasubramanian + +Authors' Addresses + + Jana Iyengar (editor) + Fastly + + Email: jri.ietf@gmail.com + + + Ian Swett (editor) + Google + + Email: ianswett@google.com diff --git a/eval/corpora/rfc/RFC 9114 - HTTP3.txt b/eval/corpora/rfc/RFC 9114 - HTTP3.txt new file mode 100644 index 00000000..67d5b0bd --- /dev/null +++ b/eval/corpora/rfc/RFC 9114 - HTTP3.txt @@ -0,0 +1,3234 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) M. Bishop, Ed. +Request for Comments: 9114 Akamai +Category: Standards Track June 2022 +ISSN: 2070-1721 + + + HTTP/3 + +Abstract + + The QUIC transport protocol has several features that are desirable + in a transport for HTTP, such as stream multiplexing, per-stream flow + control, and low-latency connection establishment. This document + describes a mapping of HTTP semantics over QUIC. This document also + identifies HTTP/2 features that are subsumed by QUIC and describes + how HTTP/2 extensions can be ported to HTTP/3. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9114. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 1.1. Prior Versions of HTTP + 1.2. Delegation to QUIC + 2. HTTP/3 Protocol Overview + 2.1. Document Organization + 2.2. Conventions and Terminology + 3. Connection Setup and Management + 3.1. Discovering an HTTP/3 Endpoint + 3.1.1. HTTP Alternative Services + 3.1.2. Other Schemes + 3.2. Connection Establishment + 3.3. Connection Reuse + 4. Expressing HTTP Semantics in HTTP/3 + 4.1. HTTP Message Framing + 4.1.1. Request Cancellation and Rejection + 4.1.2. Malformed Requests and Responses + 4.2. HTTP Fields + 4.2.1. Field Compression + 4.2.2. Header Size Constraints + 4.3. HTTP Control Data + 4.3.1. Request Pseudo-Header Fields + 4.3.2. Response Pseudo-Header Fields + 4.4. The CONNECT Method + 4.5. HTTP Upgrade + 4.6. Server Push + 5. Connection Closure + 5.1. Idle Connections + 5.2. Connection Shutdown + 5.3. Immediate Application Closure + 5.4. Transport Closure + 6. Stream Mapping and Usage + 6.1. Bidirectional Streams + 6.2. Unidirectional Streams + 6.2.1. Control Streams + 6.2.2. Push Streams + 6.2.3. Reserved Stream Types + 7. HTTP Framing Layer + 7.1. Frame Layout + 7.2. Frame Definitions + 7.2.1. DATA + 7.2.2. HEADERS + 7.2.3. CANCEL_PUSH + 7.2.4. SETTINGS + 7.2.5. PUSH_PROMISE + 7.2.6. GOAWAY + 7.2.7. MAX_PUSH_ID + 7.2.8. Reserved Frame Types + 8. Error Handling + 8.1. HTTP/3 Error Codes + 9. Extensions to HTTP/3 + 10. Security Considerations + 10.1. Server Authority + 10.2. Cross-Protocol Attacks + 10.3. Intermediary-Encapsulation Attacks + 10.4. Cacheability of Pushed Responses + 10.5. Denial-of-Service Considerations + 10.5.1. Limits on Field Section Size + 10.5.2. CONNECT Issues + 10.6. Use of Compression + 10.7. Padding and Traffic Analysis + 10.8. Frame Parsing + 10.9. Early Data + 10.10. Migration + 10.11. Privacy Considerations + 11. IANA Considerations + 11.1. Registration of HTTP/3 Identification String + 11.2. New Registries + 11.2.1. Frame Types + 11.2.2. Settings Parameters + 11.2.3. Error Codes + 11.2.4. Stream Types + 12. References + 12.1. Normative References + 12.2. Informative References + Appendix A. Considerations for Transitioning from HTTP/2 + A.1. Streams + A.2. HTTP Frame Types + A.2.1. Prioritization Differences + A.2.2. Field Compression Differences + A.2.3. Flow-Control Differences + A.2.4. Guidance for New Frame Type Definitions + A.2.5. Comparison of HTTP/2 and HTTP/3 Frame Types + A.3. HTTP/2 SETTINGS Parameters + A.4. HTTP/2 Error Codes + A.4.1. Mapping between HTTP/2 and HTTP/3 Errors + Acknowledgments + Index + Author's Address + +1. Introduction + + HTTP semantics ([HTTP]) are used for a broad range of services on the + Internet. These semantics have most commonly been used with HTTP/1.1 + and HTTP/2. HTTP/1.1 has been used over a variety of transport and + session layers, while HTTP/2 has been used primarily with TLS over + TCP. HTTP/3 supports the same semantics over a new transport + protocol: QUIC. + +1.1. Prior Versions of HTTP + + HTTP/1.1 ([HTTP/1.1]) uses whitespace-delimited text fields to convey + HTTP messages. While these exchanges are human readable, using + whitespace for message formatting leads to parsing complexity and + excessive tolerance of variant behavior. + + Because HTTP/1.1 does not include a multiplexing layer, multiple TCP + connections are often used to service requests in parallel. However, + that has a negative impact on congestion control and network + efficiency, since TCP does not share congestion control across + multiple connections. + + HTTP/2 ([HTTP/2]) introduced a binary framing and multiplexing layer + to improve latency without modifying the transport layer. However, + because the parallel nature of HTTP/2's multiplexing is not visible + to TCP's loss recovery mechanisms, a lost or reordered packet causes + all active transactions to experience a stall regardless of whether + that transaction was directly impacted by the lost packet. + +1.2. Delegation to QUIC + + The QUIC transport protocol incorporates stream multiplexing and per- + stream flow control, similar to that provided by the HTTP/2 framing + layer. By providing reliability at the stream level and congestion + control across the entire connection, QUIC has the capability to + improve the performance of HTTP compared to a TCP mapping. QUIC also + incorporates TLS 1.3 ([TLS]) at the transport layer, offering + comparable confidentiality and integrity to running TLS over TCP, + with the improved connection setup latency of TCP Fast Open ([TFO]). + + This document defines HTTP/3: a mapping of HTTP semantics over the + QUIC transport protocol, drawing heavily on the design of HTTP/2. + HTTP/3 relies on QUIC to provide confidentiality and integrity + protection of data; peer authentication; and reliable, in-order, per- + stream delivery. While delegating stream lifetime and flow-control + issues to QUIC, a binary framing similar to the HTTP/2 framing is + used on each stream. Some HTTP/2 features are subsumed by QUIC, + while other features are implemented atop QUIC. + + QUIC is described in [QUIC-TRANSPORT]. For a full description of + HTTP/2, see [HTTP/2]. + +2. HTTP/3 Protocol Overview + + HTTP/3 provides a transport for HTTP semantics using the QUIC + transport protocol and an internal framing layer similar to HTTP/2. + + Once a client knows that an HTTP/3 server exists at a certain + endpoint, it opens a QUIC connection. QUIC provides protocol + negotiation, stream-based multiplexing, and flow control. Discovery + of an HTTP/3 endpoint is described in Section 3.1. + + Within each stream, the basic unit of HTTP/3 communication is a frame + (Section 7.2). Each frame type serves a different purpose. For + example, HEADERS and DATA frames form the basis of HTTP requests and + responses (Section 4.1). Frames that apply to the entire connection + are conveyed on a dedicated control stream. + + Multiplexing of requests is performed using the QUIC stream + abstraction, which is described in Section 2 of [QUIC-TRANSPORT]. + Each request-response pair consumes a single QUIC stream. Streams + are independent of each other, so one stream that is blocked or + suffers packet loss does not prevent progress on other streams. + + Server push is an interaction mode introduced in HTTP/2 ([HTTP/2]) + that permits a server to push a request-response exchange to a client + in anticipation of the client making the indicated request. This + trades off network usage against a potential latency gain. Several + HTTP/3 frames are used to manage server push, such as PUSH_PROMISE, + MAX_PUSH_ID, and CANCEL_PUSH. + + As in HTTP/2, request and response fields are compressed for + transmission. Because HPACK ([HPACK]) relies on in-order + transmission of compressed field sections (a guarantee not provided + by QUIC), HTTP/3 replaces HPACK with QPACK ([QPACK]). QPACK uses + separate unidirectional streams to modify and track field table + state, while encoded field sections refer to the state of the table + without modifying it. + +2.1. Document Organization + + The following sections provide a detailed overview of the lifecycle + of an HTTP/3 connection: + + * "Connection Setup and Management" (Section 3) covers how an HTTP/3 + endpoint is discovered and an HTTP/3 connection is established. + + * "Expressing HTTP Semantics in HTTP/3" (Section 4) describes how + HTTP semantics are expressed using frames. + + * "Connection Closure" (Section 5) describes how HTTP/3 connections + are terminated, either gracefully or abruptly. + + The details of the wire protocol and interactions with the transport + are described in subsequent sections: + + * "Stream Mapping and Usage" (Section 6) describes the way QUIC + streams are used. + + * "HTTP Framing Layer" (Section 7) describes the frames used on most + streams. + + * "Error Handling" (Section 8) describes how error conditions are + handled and expressed, either on a particular stream or for the + connection as a whole. + + Additional resources are provided in the final sections: + + * "Extensions to HTTP/3" (Section 9) describes how new capabilities + can be added in future documents. + + * A more detailed comparison between HTTP/2 and HTTP/3 can be found + in Appendix A. + +2.2. Conventions and Terminology + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + This document uses the variable-length integer encoding from + [QUIC-TRANSPORT]. + + The following terms are used: + + abort: An abrupt termination of a connection or stream, possibly due + to an error condition. + + client: The endpoint that initiates an HTTP/3 connection. Clients + send HTTP requests and receive HTTP responses. + + connection: A transport-layer connection between two endpoints using + QUIC as the transport protocol. + + connection error: An error that affects the entire HTTP/3 + connection. + + endpoint: Either the client or server of the connection. + + frame: The smallest unit of communication on a stream in HTTP/3, + consisting of a header and a variable-length sequence of bytes + structured according to the frame type. + + Protocol elements called "frames" exist in both this document and + [QUIC-TRANSPORT]. Where frames from [QUIC-TRANSPORT] are + referenced, the frame name will be prefaced with "QUIC". For + example, "QUIC CONNECTION_CLOSE frames". References without this + preface refer to frames defined in Section 7.2. + + HTTP/3 connection: A QUIC connection where the negotiated + application protocol is HTTP/3. + + peer: An endpoint. When discussing a particular endpoint, "peer" + refers to the endpoint that is remote to the primary subject of + discussion. + + receiver: An endpoint that is receiving frames. + + sender: An endpoint that is transmitting frames. + + server: The endpoint that accepts an HTTP/3 connection. Servers + receive HTTP requests and send HTTP responses. + + stream: A bidirectional or unidirectional bytestream provided by the + QUIC transport. All streams within an HTTP/3 connection can be + considered "HTTP/3 streams", but multiple stream types are defined + within HTTP/3. + + stream error: An application-level error on the individual stream. + + The term "content" is defined in Section 6.4 of [HTTP]. + + Finally, the terms "resource", "message", "user agent", "origin + server", "gateway", "intermediary", "proxy", and "tunnel" are defined + in Section 3 of [HTTP]. + + Packet diagrams in this document use the format defined in + Section 1.3 of [QUIC-TRANSPORT] to illustrate the order and size of + fields. + +3. Connection Setup and Management + +3.1. Discovering an HTTP/3 Endpoint + + HTTP relies on the notion of an authoritative response: a response + that has been determined to be the most appropriate response for that + request given the state of the target resource at the time of + response message origination by (or at the direction of) the origin + server identified within the target URI. Locating an authoritative + server for an HTTP URI is discussed in Section 4.3 of [HTTP]. + + The "https" scheme associates authority with possession of a + certificate that the client considers to be trustworthy for the host + identified by the authority component of the URI. Upon receiving a + server certificate in the TLS handshake, the client MUST verify that + the certificate is an acceptable match for the URI's origin server + using the process described in Section 4.3.4 of [HTTP]. If the + certificate cannot be verified with respect to the URI's origin + server, the client MUST NOT consider the server authoritative for + that origin. + + A client MAY attempt access to a resource with an "https" URI by + resolving the host identifier to an IP address, establishing a QUIC + connection to that address on the indicated port (including + validation of the server certificate as described above), and sending + an HTTP/3 request message targeting the URI to the server over that + secured connection. Unless some other mechanism is used to select + HTTP/3, the token "h3" is used in the Application-Layer Protocol + Negotiation (ALPN; see [RFC7301]) extension during the TLS handshake. + + Connectivity problems (e.g., blocking UDP) can result in a failure to + establish a QUIC connection; clients SHOULD attempt to use TCP-based + versions of HTTP in this case. + + Servers MAY serve HTTP/3 on any UDP port; an alternative service + advertisement always includes an explicit port, and URIs contain + either an explicit port or a default port associated with the scheme. + +3.1.1. HTTP Alternative Services + + An HTTP origin can advertise the availability of an equivalent HTTP/3 + endpoint via the Alt-Svc HTTP response header field or the HTTP/2 + ALTSVC frame ([ALTSVC]) using the "h3" ALPN token. + + For example, an origin could indicate in an HTTP response that HTTP/3 + was available on UDP port 50781 at the same hostname by including the + following header field: + + Alt-Svc: h3=":50781" + + On receipt of an Alt-Svc record indicating HTTP/3 support, a client + MAY attempt to establish a QUIC connection to the indicated host and + port; if this connection is successful, the client can send HTTP + requests using the mapping described in this document. + +3.1.2. Other Schemes + + Although HTTP is independent of the transport protocol, the "http" + scheme associates authority with the ability to receive TCP + connections on the indicated port of whatever host is identified + within the authority component. Because HTTP/3 does not use TCP, + HTTP/3 cannot be used for direct access to the authoritative server + for a resource identified by an "http" URI. However, protocol + extensions such as [ALTSVC] permit the authoritative server to + identify other services that are also authoritative and that might be + reachable over HTTP/3. + + Prior to making requests for an origin whose scheme is not "https", + the client MUST ensure the server is willing to serve that scheme. + For origins whose scheme is "http", an experimental method to + accomplish this is described in [RFC8164]. Other mechanisms might be + defined for various schemes in the future. + +3.2. Connection Establishment + + HTTP/3 relies on QUIC version 1 as the underlying transport. The use + of other QUIC transport versions with HTTP/3 MAY be defined by future + specifications. + + QUIC version 1 uses TLS version 1.3 or greater as its handshake + protocol. HTTP/3 clients MUST support a mechanism to indicate the + target host to the server during the TLS handshake. If the server is + identified by a domain name ([DNS-TERMS]), clients MUST send the + Server Name Indication (SNI; [RFC6066]) TLS extension unless an + alternative mechanism to indicate the target host is used. + + QUIC connections are established as described in [QUIC-TRANSPORT]. + During connection establishment, HTTP/3 support is indicated by + selecting the ALPN token "h3" in the TLS handshake. Support for + other application-layer protocols MAY be offered in the same + handshake. + + While connection-level options pertaining to the core QUIC protocol + are set in the initial crypto handshake, settings specific to HTTP/3 + are conveyed in the SETTINGS frame. After the QUIC connection is + established, a SETTINGS frame MUST be sent by each endpoint as the + initial frame of their respective HTTP control stream. + +3.3. Connection Reuse + + HTTP/3 connections are persistent across multiple requests. For best + performance, it is expected that clients will not close connections + until it is determined that no further communication with a server is + necessary (for example, when a user navigates away from a particular + web page) or until the server closes the connection. + + Once a connection to a server endpoint exists, this connection MAY be + reused for requests with multiple different URI authority components. + To use an existing connection for a new origin, clients MUST validate + the certificate presented by the server for the new origin server + using the process described in Section 4.3.4 of [HTTP]. This implies + that clients will need to retain the server certificate and any + additional information needed to verify that certificate; clients + that do not do so will be unable to reuse the connection for + additional origins. + + If the certificate is not acceptable with regard to the new origin + for any reason, the connection MUST NOT be reused and a new + connection SHOULD be established for the new origin. If the reason + the certificate cannot be verified might apply to other origins + already associated with the connection, the client SHOULD revalidate + the server certificate for those origins. For instance, if + validation of a certificate fails because the certificate has expired + or been revoked, this might be used to invalidate all other origins + for which that certificate was used to establish authority. + + Clients SHOULD NOT open more than one HTTP/3 connection to a given IP + address and UDP port, where the IP address and port might be derived + from a URI, a selected alternative service ([ALTSVC]), a configured + proxy, or name resolution of any of these. A client MAY open + multiple HTTP/3 connections to the same IP address and UDP port using + different transport or TLS configurations but SHOULD avoid creating + multiple connections with the same configuration. + + Servers are encouraged to maintain open HTTP/3 connections for as + long as possible but are permitted to terminate idle connections if + necessary. When either endpoint chooses to close the HTTP/3 + connection, the terminating endpoint SHOULD first send a GOAWAY frame + (Section 5.2) so that both endpoints can reliably determine whether + previously sent frames have been processed and gracefully complete or + terminate any necessary remaining tasks. + + A server that does not wish clients to reuse HTTP/3 connections for a + particular origin can indicate that it is not authoritative for a + request by sending a 421 (Misdirected Request) status code in + response to the request; see Section 7.4 of [HTTP]. + +4. Expressing HTTP Semantics in HTTP/3 + +4.1. HTTP Message Framing + + A client sends an HTTP request on a request stream, which is a + client-initiated bidirectional QUIC stream; see Section 6.1. A + client MUST send only a single request on a given stream. A server + sends zero or more interim HTTP responses on the same stream as the + request, followed by a single final HTTP response, as detailed below. + See Section 15 of [HTTP] for a description of interim and final HTTP + responses. + + Pushed responses are sent on a server-initiated unidirectional QUIC + stream; see Section 6.2.2. A server sends zero or more interim HTTP + responses, followed by a single final HTTP response, in the same + manner as a standard response. Push is described in more detail in + Section 4.6. + + On a given stream, receipt of multiple requests or receipt of an + additional HTTP response following a final HTTP response MUST be + treated as malformed. + + An HTTP message (request or response) consists of: + + 1. the header section, including message control data, sent as a + single HEADERS frame, + + 2. optionally, the content, if present, sent as a series of DATA + frames, and + + 3. optionally, the trailer section, if present, sent as a single + HEADERS frame. + + Header and trailer sections are described in Sections 6.3 and 6.5 of + [HTTP]; the content is described in Section 6.4 of [HTTP]. + + Receipt of an invalid sequence of frames MUST be treated as a + connection error of type H3_FRAME_UNEXPECTED. In particular, a DATA + frame before any HEADERS frame, or a HEADERS or DATA frame after the + trailing HEADERS frame, is considered invalid. Other frame types, + especially unknown frame types, might be permitted subject to their + own rules; see Section 9. + + A server MAY send one or more PUSH_PROMISE frames before, after, or + interleaved with the frames of a response message. These + PUSH_PROMISE frames are not part of the response; see Section 4.6 for + more details. PUSH_PROMISE frames are not permitted on push streams; + a pushed response that includes PUSH_PROMISE frames MUST be treated + as a connection error of type H3_FRAME_UNEXPECTED. + + Frames of unknown types (Section 9), including reserved frames + (Section 7.2.8) MAY be sent on a request or push stream before, + after, or interleaved with other frames described in this section. + + The HEADERS and PUSH_PROMISE frames might reference updates to the + QPACK dynamic table. While these updates are not directly part of + the message exchange, they must be received and processed before the + message can be consumed. See Section 4.2 for more details. + + Transfer codings (see Section 7 of [HTTP/1.1]) are not defined for + HTTP/3; the Transfer-Encoding header field MUST NOT be used. + + A response MAY consist of multiple messages when and only when one or + more interim responses (1xx; see Section 15.2 of [HTTP]) precede a + final response to the same request. Interim responses do not contain + content or trailer sections. + + An HTTP request/response exchange fully consumes a client-initiated + bidirectional QUIC stream. After sending a request, a client MUST + close the stream for sending. Unless using the CONNECT method (see + Section 4.4), clients MUST NOT make stream closure dependent on + receiving a response to their request. After sending a final + response, the server MUST close the stream for sending. At this + point, the QUIC stream is fully closed. + + When a stream is closed, this indicates the end of the final HTTP + message. Because some messages are large or unbounded, endpoints + SHOULD begin processing partial HTTP messages once enough of the + message has been received to make progress. If a client-initiated + stream terminates without enough of the HTTP message to provide a + complete response, the server SHOULD abort its response stream with + the error code H3_REQUEST_INCOMPLETE. + + A server can send a complete response prior to the client sending an + entire request if the response does not depend on any portion of the + request that has not been sent and received. When the server does + not need to receive the remainder of the request, it MAY abort + reading the request stream, send a complete response, and cleanly + close the sending part of the stream. The error code H3_NO_ERROR + SHOULD be used when requesting that the client stop sending on the + request stream. Clients MUST NOT discard complete responses as a + result of having their request terminated abruptly, though clients + can always discard responses at their discretion for other reasons. + If the server sends a partial or complete response but does not abort + reading the request, clients SHOULD continue sending the content of + the request and close the stream normally. + +4.1.1. Request Cancellation and Rejection + + Once a request stream has been opened, the request MAY be cancelled + by either endpoint. Clients cancel requests if the response is no + longer of interest; servers cancel requests if they are unable to or + choose not to respond. When possible, it is RECOMMENDED that servers + send an HTTP response with an appropriate status code rather than + cancelling a request it has already begun processing. + + Implementations SHOULD cancel requests by abruptly terminating any + directions of a stream that are still open. To do so, an + implementation resets the sending parts of streams and aborts reading + on the receiving parts of streams; see Section 2.4 of + [QUIC-TRANSPORT]. + + When the server cancels a request without performing any application + processing, the request is considered "rejected". The server SHOULD + abort its response stream with the error code H3_REQUEST_REJECTED. + In this context, "processed" means that some data from the stream was + passed to some higher layer of software that might have taken some + action as a result. The client can treat requests rejected by the + server as though they had never been sent at all, thereby allowing + them to be retried later. + + Servers MUST NOT use the H3_REQUEST_REJECTED error code for requests + that were partially or fully processed. When a server abandons a + response after partial processing, it SHOULD abort its response + stream with the error code H3_REQUEST_CANCELLED. + + Client SHOULD use the error code H3_REQUEST_CANCELLED to cancel + requests. Upon receipt of this error code, a server MAY abruptly + terminate the response using the error code H3_REQUEST_REJECTED if no + processing was performed. Clients MUST NOT use the + H3_REQUEST_REJECTED error code, except when a server has requested + closure of the request stream with this error code. + + If a stream is cancelled after receiving a complete response, the + client MAY ignore the cancellation and use the response. However, if + a stream is cancelled after receiving a partial response, the + response SHOULD NOT be used. Only idempotent actions such as GET, + PUT, or DELETE can be safely retried; a client SHOULD NOT + automatically retry a request with a non-idempotent method unless it + has some means to know that the request semantics are idempotent + independent of the method or some means to detect that the original + request was never applied. See Section 9.2.2 of [HTTP] for more + details. + +4.1.2. Malformed Requests and Responses + + A malformed request or response is one that is an otherwise valid + sequence of frames but is invalid due to: + + * the presence of prohibited fields or pseudo-header fields, + + * the absence of mandatory pseudo-header fields, + + * invalid values for pseudo-header fields, + + * pseudo-header fields after fields, + + * an invalid sequence of HTTP messages, + + * the inclusion of uppercase field names, or + + * the inclusion of invalid characters in field names or values. + + A request or response that is defined as having content when it + contains a Content-Length header field (Section 8.6 of [HTTP]) is + malformed if the value of the Content-Length header field does not + equal the sum of the DATA frame lengths received. A response that is + defined as never having content, even when a Content-Length is + present, can have a non-zero Content-Length header field even though + no content is included in DATA frames. + + Intermediaries that process HTTP requests or responses (i.e., any + intermediary not acting as a tunnel) MUST NOT forward a malformed + request or response. Malformed requests or responses that are + detected MUST be treated as a stream error of type H3_MESSAGE_ERROR. + + For malformed requests, a server MAY send an HTTP response indicating + the error prior to closing or resetting the stream. Clients MUST NOT + accept a malformed response. Note that these requirements are + intended to protect against several types of common attacks against + HTTP; they are deliberately strict because being permissive can + expose implementations to these vulnerabilities. + +4.2. HTTP Fields + + HTTP messages carry metadata as a series of key-value pairs called + "HTTP fields"; see Sections 6.3 and 6.5 of [HTTP]. For a listing of + registered HTTP fields, see the "Hypertext Transfer Protocol (HTTP) + Field Name Registry" maintained at <https://www.iana.org/assignments/ + http-fields/>. Like HTTP/2, HTTP/3 has additional considerations + related to the use of characters in field names, the Connection + header field, and pseudo-header fields. + + Field names are strings containing a subset of ASCII characters. + Properties of HTTP field names and values are discussed in more + detail in Section 5.1 of [HTTP]. Characters in field names MUST be + converted to lowercase prior to their encoding. A request or + response containing uppercase characters in field names MUST be + treated as malformed. + + HTTP/3 does not use the Connection header field to indicate + connection-specific fields; in this protocol, connection-specific + metadata is conveyed by other means. An endpoint MUST NOT generate + an HTTP/3 field section containing connection-specific fields; any + message containing connection-specific fields MUST be treated as + malformed. + + The only exception to this is the TE header field, which MAY be + present in an HTTP/3 request header; when it is, it MUST NOT contain + any value other than "trailers". + + An intermediary transforming an HTTP/1.x message to HTTP/3 MUST + remove connection-specific header fields as discussed in + Section 7.6.1 of [HTTP], or their messages will be treated by other + HTTP/3 endpoints as malformed. + +4.2.1. Field Compression + + [QPACK] describes a variation of HPACK that gives an encoder some + control over how much head-of-line blocking can be caused by + compression. This allows an encoder to balance compression + efficiency with latency. HTTP/3 uses QPACK to compress header and + trailer sections, including the control data present in the header + section. + + To allow for better compression efficiency, the Cookie header field + ([COOKIES]) MAY be split into separate field lines, each with one or + more cookie-pairs, before compression. If a decompressed field + section contains multiple cookie field lines, these MUST be + concatenated into a single byte string using the two-byte delimiter + of "; " (ASCII 0x3b, 0x20) before being passed into a context other + than HTTP/2 or HTTP/3, such as an HTTP/1.1 connection, or a generic + HTTP server application. + +4.2.2. Header Size Constraints + + An HTTP/3 implementation MAY impose a limit on the maximum size of + the message header it will accept on an individual HTTP message. A + server that receives a larger header section than it is willing to + handle can send an HTTP 431 (Request Header Fields Too Large) status + code ([RFC6585]). A client can discard responses that it cannot + process. The size of a field list is calculated based on the + uncompressed size of fields, including the length of the name and + value in bytes plus an overhead of 32 bytes for each field. + + If an implementation wishes to advise its peer of this limit, it can + be conveyed as a number of bytes in the + SETTINGS_MAX_FIELD_SECTION_SIZE parameter. An implementation that + has received this parameter SHOULD NOT send an HTTP message header + that exceeds the indicated size, as the peer will likely refuse to + process it. However, an HTTP message can traverse one or more + intermediaries before reaching the origin server; see Section 3.7 of + [HTTP]. Because this limit is applied separately by each + implementation that processes the message, messages below this limit + are not guaranteed to be accepted. + +4.3. HTTP Control Data + + Like HTTP/2, HTTP/3 employs a series of pseudo-header fields, where + the field name begins with the : character (ASCII 0x3a). These + pseudo-header fields convey message control data; see Section 6.2 of + [HTTP]. + + Pseudo-header fields are not HTTP fields. Endpoints MUST NOT + generate pseudo-header fields other than those defined in this + document. However, an extension could negotiate a modification of + this restriction; see Section 9. + + Pseudo-header fields are only valid in the context in which they are + defined. Pseudo-header fields defined for requests MUST NOT appear + in responses; pseudo-header fields defined for responses MUST NOT + appear in requests. Pseudo-header fields MUST NOT appear in trailer + sections. Endpoints MUST treat a request or response that contains + undefined or invalid pseudo-header fields as malformed. + + All pseudo-header fields MUST appear in the header section before + regular header fields. Any request or response that contains a + pseudo-header field that appears in a header section after a regular + header field MUST be treated as malformed. + +4.3.1. Request Pseudo-Header Fields + + The following pseudo-header fields are defined for requests: + + ":method": Contains the HTTP method (Section 9 of [HTTP]) + + ":scheme": Contains the scheme portion of the target URI + (Section 3.1 of [URI]). + + The :scheme pseudo-header is not restricted to URIs with scheme + "http" and "https". A proxy or gateway can translate requests for + non-HTTP schemes, enabling the use of HTTP to interact with non- + HTTP services. + + See Section 3.1.2 for guidance on using a scheme other than + "https". + + ":authority": Contains the authority portion of the target URI + (Section 3.2 of [URI]). The authority MUST NOT include the + deprecated userinfo subcomponent for URIs of scheme "http" or + "https". + + To ensure that the HTTP/1.1 request line can be reproduced + accurately, this pseudo-header field MUST be omitted when + translating from an HTTP/1.1 request that has a request target in + a method-specific form; see Section 7.1 of [HTTP]. Clients that + generate HTTP/3 requests directly SHOULD use the :authority + pseudo-header field instead of the Host header field. An + intermediary that converts an HTTP/3 request to HTTP/1.1 MUST + create a Host field if one is not present in a request by copying + the value of the :authority pseudo-header field. + + ":path": Contains the path and query parts of the target URI (the + "path-absolute" production and optionally a ? character (ASCII + 0x3f) followed by the "query" production; see Sections 3.3 and 3.4 + of [URI]. + + This pseudo-header field MUST NOT be empty for "http" or "https" + URIs; "http" or "https" URIs that do not contain a path component + MUST include a value of / (ASCII 0x2f). An OPTIONS request that + does not include a path component includes the value * (ASCII + 0x2a) for the :path pseudo-header field; see Section 7.1 of + [HTTP]. + + All HTTP/3 requests MUST include exactly one value for the :method, + :scheme, and :path pseudo-header fields, unless the request is a + CONNECT request; see Section 4.4. + + If the :scheme pseudo-header field identifies a scheme that has a + mandatory authority component (including "http" and "https"), the + request MUST contain either an :authority pseudo-header field or a + Host header field. If these fields are present, they MUST NOT be + empty. If both fields are present, they MUST contain the same value. + If the scheme does not have a mandatory authority component and none + is provided in the request target, the request MUST NOT contain the + :authority pseudo-header or Host header fields. + + An HTTP request that omits mandatory pseudo-header fields or contains + invalid values for those pseudo-header fields is malformed. + + HTTP/3 does not define a way to carry the version identifier that is + included in the HTTP/1.1 request line. HTTP/3 requests implicitly + have a protocol version of "3.0". + +4.3.2. Response Pseudo-Header Fields + + For responses, a single ":status" pseudo-header field is defined that + carries the HTTP status code; see Section 15 of [HTTP]. This pseudo- + header field MUST be included in all responses; otherwise, the + response is malformed (see Section 4.1.2). + + HTTP/3 does not define a way to carry the version or reason phrase + that is included in an HTTP/1.1 status line. HTTP/3 responses + implicitly have a protocol version of "3.0". + +4.4. The CONNECT Method + + The CONNECT method requests that the recipient establish a tunnel to + the destination origin server identified by the request-target; see + Section 9.3.6 of [HTTP]. It is primarily used with HTTP proxies to + establish a TLS session with an origin server for the purposes of + interacting with "https" resources. + + In HTTP/1.x, CONNECT is used to convert an entire HTTP connection + into a tunnel to a remote host. In HTTP/2 and HTTP/3, the CONNECT + method is used to establish a tunnel over a single stream. + + A CONNECT request MUST be constructed as follows: + + * The :method pseudo-header field is set to "CONNECT" + + * The :scheme and :path pseudo-header fields are omitted + + * The :authority pseudo-header field contains the host and port to + connect to (equivalent to the authority-form of the request-target + of CONNECT requests; see Section 7.1 of [HTTP]). + + The request stream remains open at the end of the request to carry + the data to be transferred. A CONNECT request that does not conform + to these restrictions is malformed. + + A proxy that supports CONNECT establishes a TCP connection + ([RFC0793]) to the server identified in the :authority pseudo-header + field. Once this connection is successfully established, the proxy + sends a HEADERS frame containing a 2xx series status code to the + client, as defined in Section 15.3 of [HTTP]. + + All DATA frames on the stream correspond to data sent or received on + the TCP connection. The payload of any DATA frame sent by the client + is transmitted by the proxy to the TCP server; data received from the + TCP server is packaged into DATA frames by the proxy. Note that the + size and number of TCP segments is not guaranteed to map predictably + to the size and number of HTTP DATA or QUIC STREAM frames. + + Once the CONNECT method has completed, only DATA frames are permitted + to be sent on the stream. Extension frames MAY be used if + specifically permitted by the definition of the extension. Receipt + of any other known frame type MUST be treated as a connection error + of type H3_FRAME_UNEXPECTED. + + The TCP connection can be closed by either peer. When the client + ends the request stream (that is, the receive stream at the proxy + enters the "Data Recvd" state), the proxy will set the FIN bit on its + connection to the TCP server. When the proxy receives a packet with + the FIN bit set, it will close the send stream that it sends to the + client. TCP connections that remain half closed in a single + direction are not invalid, but are often handled poorly by servers, + so clients SHOULD NOT close a stream for sending while they still + expect to receive data from the target of the CONNECT. + + A TCP connection error is signaled by abruptly terminating the + stream. A proxy treats any error in the TCP connection, which + includes receiving a TCP segment with the RST bit set, as a stream + error of type H3_CONNECT_ERROR. + + Correspondingly, if a proxy detects an error with the stream or the + QUIC connection, it MUST close the TCP connection. If the proxy + detects that the client has reset the stream or aborted reading from + the stream, it MUST close the TCP connection. If the stream is reset + or reading is aborted by the client, a proxy SHOULD perform the same + operation on the other direction in order to ensure that both + directions of the stream are cancelled. In all these cases, if the + underlying TCP implementation permits it, the proxy SHOULD send a TCP + segment with the RST bit set. + + Since CONNECT creates a tunnel to an arbitrary server, proxies that + support CONNECT SHOULD restrict its use to a set of known ports or a + list of safe request targets; see Section 9.3.6 of [HTTP] for more + details. + +4.5. HTTP Upgrade + + HTTP/3 does not support the HTTP Upgrade mechanism (Section 7.8 of + [HTTP]) or the 101 (Switching Protocols) informational status code + (Section 15.2.2 of [HTTP]). + +4.6. Server Push + + Server push is an interaction mode that permits a server to push a + request-response exchange to a client in anticipation of the client + making the indicated request. This trades off network usage against + a potential latency gain. HTTP/3 server push is similar to what is + described in Section 8.2 of [HTTP/2], but it uses different + mechanisms. + + Each server push is assigned a unique push ID by the server. The + push ID is used to refer to the push in various contexts throughout + the lifetime of the HTTP/3 connection. + + The push ID space begins at zero and ends at a maximum value set by + the MAX_PUSH_ID frame. In particular, a server is not able to push + until after the client sends a MAX_PUSH_ID frame. A client sends + MAX_PUSH_ID frames to control the number of pushes that a server can + promise. A server SHOULD use push IDs sequentially, beginning from + zero. A client MUST treat receipt of a push stream as a connection + error of type H3_ID_ERROR when no MAX_PUSH_ID frame has been sent or + when the stream references a push ID that is greater than the maximum + push ID. + + The push ID is used in one or more PUSH_PROMISE frames that carry the + control data and header fields of the request message. These frames + are sent on the request stream that generated the push. This allows + the server push to be associated with a client request. When the + same push ID is promised on multiple request streams, the + decompressed request field sections MUST contain the same fields in + the same order, and both the name and the value in each field MUST be + identical. + + The push ID is then included with the push stream that ultimately + fulfills those promises. The push stream identifies the push ID of + the promise that it fulfills, then contains a response to the + promised request as described in Section 4.1. + + Finally, the push ID can be used in CANCEL_PUSH frames; see + Section 7.2.3. Clients use this frame to indicate they do not wish + to receive a promised resource. Servers use this frame to indicate + they will not be fulfilling a previous promise. + + Not all requests can be pushed. A server MAY push requests that have + the following properties: + + * cacheable; see Section 9.2.3 of [HTTP] + + * safe; see Section 9.2.1 of [HTTP] + + * does not include request content or a trailer section + + The server MUST include a value in the :authority pseudo-header field + for which the server is authoritative. If the client has not yet + validated the connection for the origin indicated by the pushed + request, it MUST perform the same verification process it would do + before sending a request for that origin on the connection; see + Section 3.3. If this verification fails, the client MUST NOT + consider the server authoritative for that origin. + + Clients SHOULD send a CANCEL_PUSH frame upon receipt of a + PUSH_PROMISE frame carrying a request that is not cacheable, is not + known to be safe, that indicates the presence of request content, or + for which it does not consider the server authoritative. Any + corresponding responses MUST NOT be used or cached. + + Each pushed response is associated with one or more client requests. + The push is associated with the request stream on which the + PUSH_PROMISE frame was received. The same server push can be + associated with additional client requests using a PUSH_PROMISE frame + with the same push ID on multiple request streams. These + associations do not affect the operation of the protocol, but they + MAY be considered by user agents when deciding how to use pushed + resources. + + Ordering of a PUSH_PROMISE frame in relation to certain parts of the + response is important. The server SHOULD send PUSH_PROMISE frames + prior to sending HEADERS or DATA frames that reference the promised + responses. This reduces the chance that a client requests a resource + that will be pushed by the server. + + Due to reordering, push stream data can arrive before the + corresponding PUSH_PROMISE frame. When a client receives a new push + stream with an as-yet-unknown push ID, both the associated client + request and the pushed request header fields are unknown. The client + can buffer the stream data in expectation of the matching + PUSH_PROMISE. The client can use stream flow control (Section 4.1 of + [QUIC-TRANSPORT]) to limit the amount of data a server may commit to + the pushed stream. Clients SHOULD abort reading and discard data + already read from push streams if no corresponding PUSH_PROMISE frame + is processed in a reasonable amount of time. + + Push stream data can also arrive after a client has cancelled a push. + In this case, the client can abort reading the stream with an error + code of H3_REQUEST_CANCELLED. This asks the server not to transfer + additional data and indicates that it will be discarded upon receipt. + + Pushed responses that are cacheable (see Section 3 of [HTTP-CACHING]) + can be stored by the client, if it implements an HTTP cache. Pushed + responses are considered successfully validated on the origin server + (e.g., if the "no-cache" cache response directive is present; see + Section 5.2.2.4 of [HTTP-CACHING]) at the time the pushed response is + received. + + Pushed responses that are not cacheable MUST NOT be stored by any + HTTP cache. They MAY be made available to the application + separately. + +5. Connection Closure + + Once established, an HTTP/3 connection can be used for many requests + and responses over time until the connection is closed. Connection + closure can happen in any of several different ways. + +5.1. Idle Connections + + Each QUIC endpoint declares an idle timeout during the handshake. If + the QUIC connection remains idle (no packets received) for longer + than this duration, the peer will assume that the connection has been + closed. HTTP/3 implementations will need to open a new HTTP/3 + connection for new requests if the existing connection has been idle + for longer than the idle timeout negotiated during the QUIC + handshake, and they SHOULD do so if approaching the idle timeout; see + Section 10.1 of [QUIC-TRANSPORT]. + + HTTP clients are expected to request that the transport keep + connections open while there are responses outstanding for requests + or server pushes, as described in Section 10.1.2 of [QUIC-TRANSPORT]. + If the client is not expecting a response from the server, allowing + an idle connection to time out is preferred over expending effort + maintaining a connection that might not be needed. A gateway MAY + maintain connections in anticipation of need rather than incur the + latency cost of connection establishment to servers. Servers SHOULD + NOT actively keep connections open. + +5.2. Connection Shutdown + + Even when a connection is not idle, either endpoint can decide to + stop using the connection and initiate a graceful connection close. + Endpoints initiate the graceful shutdown of an HTTP/3 connection by + sending a GOAWAY frame. The GOAWAY frame contains an identifier that + indicates to the receiver the range of requests or pushes that were + or might be processed in this connection. The server sends a client- + initiated bidirectional stream ID; the client sends a push ID. + Requests or pushes with the indicated identifier or greater are + rejected (Section 4.1.1) by the sender of the GOAWAY. This + identifier MAY be zero if no requests or pushes were processed. + + The information in the GOAWAY frame enables a client and server to + agree on which requests or pushes were accepted prior to the shutdown + of the HTTP/3 connection. Upon sending a GOAWAY frame, the endpoint + SHOULD explicitly cancel (see Sections 4.1.1 and 7.2.3) any requests + or pushes that have identifiers greater than or equal to the one + indicated, in order to clean up transport state for the affected + streams. The endpoint SHOULD continue to do so as more requests or + pushes arrive. + + Endpoints MUST NOT initiate new requests or promise new pushes on the + connection after receipt of a GOAWAY frame from the peer. Clients + MAY establish a new connection to send additional requests. + + Some requests or pushes might already be in transit: + + * Upon receipt of a GOAWAY frame, if the client has already sent + requests with a stream ID greater than or equal to the identifier + contained in the GOAWAY frame, those requests will not be + processed. Clients can safely retry unprocessed requests on a + different HTTP connection. A client that is unable to retry + requests loses all requests that are in flight when the server + closes the connection. + + Requests on stream IDs less than the stream ID in a GOAWAY frame + from the server might have been processed; their status cannot be + known until a response is received, the stream is reset + individually, another GOAWAY is received with a lower stream ID + than that of the request in question, or the connection + terminates. + + Servers MAY reject individual requests on streams below the + indicated ID if these requests were not processed. + + * If a server receives a GOAWAY frame after having promised pushes + with a push ID greater than or equal to the identifier contained + in the GOAWAY frame, those pushes will not be accepted. + + Servers SHOULD send a GOAWAY frame when the closing of a connection + is known in advance, even if the advance notice is small, so that the + remote peer can know whether or not a request has been partially + processed. For example, if an HTTP client sends a POST at the same + time that a server closes a QUIC connection, the client cannot know + if the server started to process that POST request if the server does + not send a GOAWAY frame to indicate what streams it might have acted + on. + + An endpoint MAY send multiple GOAWAY frames indicating different + identifiers, but the identifier in each frame MUST NOT be greater + than the identifier in any previous frame, since clients might + already have retried unprocessed requests on another HTTP connection. + Receiving a GOAWAY containing a larger identifier than previously + received MUST be treated as a connection error of type H3_ID_ERROR. + + An endpoint that is attempting to gracefully shut down a connection + can send a GOAWAY frame with a value set to the maximum possible + value (2^62-4 for servers, 2^62-1 for clients). This ensures that + the peer stops creating new requests or pushes. After allowing time + for any in-flight requests or pushes to arrive, the endpoint can send + another GOAWAY frame indicating which requests or pushes it might + accept before the end of the connection. This ensures that a + connection can be cleanly shut down without losing requests. + + A client has more flexibility in the value it chooses for the Push ID + field in a GOAWAY that it sends. A value of 2^62-1 indicates that + the server can continue fulfilling pushes that have already been + promised. A smaller value indicates the client will reject pushes + with push IDs greater than or equal to this value. Like the server, + the client MAY send subsequent GOAWAY frames so long as the specified + push ID is no greater than any previously sent value. + + Even when a GOAWAY indicates that a given request or push will not be + processed or accepted upon receipt, the underlying transport + resources still exist. The endpoint that initiated these requests + can cancel them to clean up transport state. + + Once all accepted requests and pushes have been processed, the + endpoint can permit the connection to become idle, or it MAY initiate + an immediate closure of the connection. An endpoint that completes a + graceful shutdown SHOULD use the H3_NO_ERROR error code when closing + the connection. + + If a client has consumed all available bidirectional stream IDs with + requests, the server need not send a GOAWAY frame, since the client + is unable to make further requests. + +5.3. Immediate Application Closure + + An HTTP/3 implementation can immediately close the QUIC connection at + any time. This results in sending a QUIC CONNECTION_CLOSE frame to + the peer indicating that the application layer has terminated the + connection. The application error code in this frame indicates to + the peer why the connection is being closed. See Section 8 for error + codes that can be used when closing a connection in HTTP/3. + + Before closing the connection, a GOAWAY frame MAY be sent to allow + the client to retry some requests. Including the GOAWAY frame in the + same packet as the QUIC CONNECTION_CLOSE frame improves the chances + of the frame being received by clients. + + If there are open streams that have not been explicitly closed, they + are implicitly closed when the connection is closed; see Section 10.2 + of [QUIC-TRANSPORT]. + +5.4. Transport Closure + + For various reasons, the QUIC transport could indicate to the + application layer that the connection has terminated. This might be + due to an explicit closure by the peer, a transport-level error, or a + change in network topology that interrupts connectivity. + + If a connection terminates without a GOAWAY frame, clients MUST + assume that any request that was sent, whether in whole or in part, + might have been processed. + +6. Stream Mapping and Usage + + A QUIC stream provides reliable in-order delivery of bytes, but makes + no guarantees about order of delivery with regard to bytes on other + streams. In version 1 of QUIC, the stream data containing HTTP + frames is carried by QUIC STREAM frames, but this framing is + invisible to the HTTP framing layer. The transport layer buffers and + orders received stream data, exposing a reliable byte stream to the + application. Although QUIC permits out-of-order delivery within a + stream, HTTP/3 does not make use of this feature. + + QUIC streams can be either unidirectional, carrying data only from + initiator to receiver, or bidirectional, carrying data in both + directions. Streams can be initiated by either the client or the + server. For more detail on QUIC streams, see Section 2 of + [QUIC-TRANSPORT]. + + When HTTP fields and data are sent over QUIC, the QUIC layer handles + most of the stream management. HTTP does not need to do any separate + multiplexing when using QUIC: data sent over a QUIC stream always + maps to a particular HTTP transaction or to the entire HTTP/3 + connection context. + +6.1. Bidirectional Streams + + All client-initiated bidirectional streams are used for HTTP requests + and responses. A bidirectional stream ensures that the response can + be readily correlated with the request. These streams are referred + to as request streams. + + This means that the client's first request occurs on QUIC stream 0, + with subsequent requests on streams 4, 8, and so on. In order to + permit these streams to open, an HTTP/3 server SHOULD configure non- + zero minimum values for the number of permitted streams and the + initial stream flow-control window. So as to not unnecessarily limit + parallelism, at least 100 request streams SHOULD be permitted at a + time. + + HTTP/3 does not use server-initiated bidirectional streams, though an + extension could define a use for these streams. Clients MUST treat + receipt of a server-initiated bidirectional stream as a connection + error of type H3_STREAM_CREATION_ERROR unless such an extension has + been negotiated. + +6.2. Unidirectional Streams + + Unidirectional streams, in either direction, are used for a range of + purposes. The purpose is indicated by a stream type, which is sent + as a variable-length integer at the start of the stream. The format + and structure of data that follows this integer is determined by the + stream type. + + Unidirectional Stream Header { + Stream Type (i), + } + + Figure 1: Unidirectional Stream Header + + Two stream types are defined in this document: control streams + (Section 6.2.1) and push streams (Section 6.2.2). [QPACK] defines + two additional stream types. Other stream types can be defined by + extensions to HTTP/3; see Section 9 for more details. Some stream + types are reserved (Section 6.2.3). + + The performance of HTTP/3 connections in the early phase of their + lifetime is sensitive to the creation and exchange of data on + unidirectional streams. Endpoints that excessively restrict the + number of streams or the flow-control window of these streams will + increase the chance that the remote peer reaches the limit early and + becomes blocked. In particular, implementations should consider that + remote peers may wish to exercise reserved stream behavior + (Section 6.2.3) with some of the unidirectional streams they are + permitted to use. + + Each endpoint needs to create at least one unidirectional stream for + the HTTP control stream. QPACK requires two additional + unidirectional streams, and other extensions might require further + streams. Therefore, the transport parameters sent by both clients + and servers MUST allow the peer to create at least three + unidirectional streams. These transport parameters SHOULD also + provide at least 1,024 bytes of flow-control credit to each + unidirectional stream. + + Note that an endpoint is not required to grant additional credits to + create more unidirectional streams if its peer consumes all the + initial credits before creating the critical unidirectional streams. + Endpoints SHOULD create the HTTP control stream as well as the + unidirectional streams required by mandatory extensions (such as the + QPACK encoder and decoder streams) first, and then create additional + streams as allowed by their peer. + + If the stream header indicates a stream type that is not supported by + the recipient, the remainder of the stream cannot be consumed as the + semantics are unknown. Recipients of unknown stream types MUST + either abort reading of the stream or discard incoming data without + further processing. If reading is aborted, the recipient SHOULD use + the H3_STREAM_CREATION_ERROR error code or a reserved error code + (Section 8.1). The recipient MUST NOT consider unknown stream types + to be a connection error of any kind. + + As certain stream types can affect connection state, a recipient + SHOULD NOT discard data from incoming unidirectional streams prior to + reading the stream type. + + Implementations MAY send stream types before knowing whether the peer + supports them. However, stream types that could modify the state or + semantics of existing protocol components, including QPACK or other + extensions, MUST NOT be sent until the peer is known to support them. + + A sender can close or reset a unidirectional stream unless otherwise + specified. A receiver MUST tolerate unidirectional streams being + closed or reset prior to the reception of the unidirectional stream + header. + +6.2.1. Control Streams + + A control stream is indicated by a stream type of 0x00. Data on this + stream consists of HTTP/3 frames, as defined in Section 7.2. + + Each side MUST initiate a single control stream at the beginning of + the connection and send its SETTINGS frame as the first frame on this + stream. If the first frame of the control stream is any other frame + type, this MUST be treated as a connection error of type + H3_MISSING_SETTINGS. Only one control stream per peer is permitted; + receipt of a second stream claiming to be a control stream MUST be + treated as a connection error of type H3_STREAM_CREATION_ERROR. The + sender MUST NOT close the control stream, and the receiver MUST NOT + request that the sender close the control stream. If either control + stream is closed at any point, this MUST be treated as a connection + error of type H3_CLOSED_CRITICAL_STREAM. Connection errors are + described in Section 8. + + Because the contents of the control stream are used to manage the + behavior of other streams, endpoints SHOULD provide enough flow- + control credit to keep the peer's control stream from becoming + blocked. + + A pair of unidirectional streams is used rather than a single + bidirectional stream. This allows either peer to send data as soon + as it is able. Depending on whether 0-RTT is available on the QUIC + connection, either client or server might be able to send stream data + first. + +6.2.2. Push Streams + + Server push is an optional feature introduced in HTTP/2 that allows a + server to initiate a response before a request has been made. See + Section 4.6 for more details. + + A push stream is indicated by a stream type of 0x01, followed by the + push ID of the promise that it fulfills, encoded as a variable-length + integer. The remaining data on this stream consists of HTTP/3 + frames, as defined in Section 7.2, and fulfills a promised server + push by zero or more interim HTTP responses followed by a single + final HTTP response, as defined in Section 4.1. Server push and push + IDs are described in Section 4.6. + + Only servers can push; if a server receives a client-initiated push + stream, this MUST be treated as a connection error of type + H3_STREAM_CREATION_ERROR. + + Push Stream Header { + Stream Type (i) = 0x01, + Push ID (i), + } + + Figure 2: Push Stream Header + + A client SHOULD NOT abort reading on a push stream prior to reading + the push stream header, as this could lead to disagreement between + client and server on which push IDs have already been consumed. + + Each push ID MUST only be used once in a push stream header. If a + client detects that a push stream header includes a push ID that was + used in another push stream header, the client MUST treat this as a + connection error of type H3_ID_ERROR. + +6.2.3. Reserved Stream Types + + Stream types of the format 0x1f * N + 0x21 for non-negative integer + values of N are reserved to exercise the requirement that unknown + types be ignored. These streams have no semantics, and they can be + sent when application-layer padding is desired. They MAY also be + sent on connections where no data is currently being transferred. + Endpoints MUST NOT consider these streams to have any meaning upon + receipt. + + The payload and length of the stream are selected in any manner the + sending implementation chooses. When sending a reserved stream type, + the implementation MAY either terminate the stream cleanly or reset + it. When resetting the stream, either the H3_NO_ERROR error code or + a reserved error code (Section 8.1) SHOULD be used. + +7. HTTP Framing Layer + + HTTP frames are carried on QUIC streams, as described in Section 6. + HTTP/3 defines three stream types: control stream, request stream, + and push stream. This section describes HTTP/3 frame formats and + their permitted stream types; see Table 1 for an overview. A + comparison between HTTP/2 and HTTP/3 frames is provided in + Appendix A.2. + + +==============+================+================+========+=========+ + | Frame | Control Stream | Request | Push | Section | + | | | Stream | Stream | | + +==============+================+================+========+=========+ + | DATA | No | Yes | Yes | Section | + | | | | | 7.2.1 | + +--------------+----------------+----------------+--------+---------+ + | HEADERS | No | Yes | Yes | Section | + | | | | | 7.2.2 | + +--------------+----------------+----------------+--------+---------+ + | CANCEL_PUSH | Yes | No | No | Section | + | | | | | 7.2.3 | + +--------------+----------------+----------------+--------+---------+ + | SETTINGS | Yes (1) | No | No | Section | + | | | | | 7.2.4 | + +--------------+----------------+----------------+--------+---------+ + | PUSH_PROMISE | No | Yes | No | Section | + | | | | | 7.2.5 | + +--------------+----------------+----------------+--------+---------+ + | GOAWAY | Yes | No | No | Section | + | | | | | 7.2.6 | + +--------------+----------------+----------------+--------+---------+ + | MAX_PUSH_ID | Yes | No | No | Section | + | | | | | 7.2.7 | + +--------------+----------------+----------------+--------+---------+ + | Reserved | Yes | Yes | Yes | Section | + | | | | | 7.2.8 | + +--------------+----------------+----------------+--------+---------+ + + Table 1: HTTP/3 Frames and Stream Type Overview + + The SETTINGS frame can only occur as the first frame of a Control + stream; this is indicated in Table 1 with a (1). Specific guidance + is provided in the relevant section. + + Note that, unlike QUIC frames, HTTP/3 frames can span multiple + packets. + +7.1. Frame Layout + + All frames have the following format: + + HTTP/3 Frame Format { + Type (i), + Length (i), + Frame Payload (..), + } + + Figure 3: HTTP/3 Frame Format + + A frame includes the following fields: + + Type: A variable-length integer that identifies the frame type. + + Length: A variable-length integer that describes the length in bytes + of the Frame Payload. + + Frame Payload: A payload, the semantics of which are determined by + the Type field. + + Each frame's payload MUST contain exactly the fields identified in + its description. A frame payload that contains additional bytes + after the identified fields or a frame payload that terminates before + the end of the identified fields MUST be treated as a connection + error of type H3_FRAME_ERROR. In particular, redundant length + encodings MUST be verified to be self-consistent; see Section 10.8. + + When a stream terminates cleanly, if the last frame on the stream was + truncated, this MUST be treated as a connection error of type + H3_FRAME_ERROR. Streams that terminate abruptly may be reset at any + point in a frame. + +7.2. Frame Definitions + +7.2.1. DATA + + DATA frames (type=0x00) convey arbitrary, variable-length sequences + of bytes associated with HTTP request or response content. + + DATA frames MUST be associated with an HTTP request or response. If + a DATA frame is received on a control stream, the recipient MUST + respond with a connection error of type H3_FRAME_UNEXPECTED. + + DATA Frame { + Type (i) = 0x00, + Length (i), + Data (..), + } + + Figure 4: DATA Frame + +7.2.2. HEADERS + + The HEADERS frame (type=0x01) is used to carry an HTTP field section + that is encoded using QPACK. See [QPACK] for more details. + + HEADERS Frame { + Type (i) = 0x01, + Length (i), + Encoded Field Section (..), + } + + Figure 5: HEADERS Frame + + HEADERS frames can only be sent on request streams or push streams. + If a HEADERS frame is received on a control stream, the recipient + MUST respond with a connection error of type H3_FRAME_UNEXPECTED. + +7.2.3. CANCEL_PUSH + + The CANCEL_PUSH frame (type=0x03) is used to request cancellation of + a server push prior to the push stream being received. The + CANCEL_PUSH frame identifies a server push by push ID (see + Section 4.6), encoded as a variable-length integer. + + When a client sends a CANCEL_PUSH frame, it is indicating that it + does not wish to receive the promised resource. The server SHOULD + abort sending the resource, but the mechanism to do so depends on the + state of the corresponding push stream. If the server has not yet + created a push stream, it does not create one. If the push stream is + open, the server SHOULD abruptly terminate that stream. If the push + stream has already ended, the server MAY still abruptly terminate the + stream or MAY take no action. + + A server sends a CANCEL_PUSH frame to indicate that it will not be + fulfilling a promise that was previously sent. The client cannot + expect the corresponding promise to be fulfilled, unless it has + already received and processed the promised response. Regardless of + whether a push stream has been opened, a server SHOULD send a + CANCEL_PUSH frame when it determines that promise will not be + fulfilled. If a stream has already been opened, the server can abort + sending on the stream with an error code of H3_REQUEST_CANCELLED. + + Sending a CANCEL_PUSH frame has no direct effect on the state of + existing push streams. A client SHOULD NOT send a CANCEL_PUSH frame + when it has already received a corresponding push stream. A push + stream could arrive after a client has sent a CANCEL_PUSH frame, + because a server might not have processed the CANCEL_PUSH. The + client SHOULD abort reading the stream with an error code of + H3_REQUEST_CANCELLED. + + A CANCEL_PUSH frame is sent on the control stream. Receiving a + CANCEL_PUSH frame on a stream other than the control stream MUST be + treated as a connection error of type H3_FRAME_UNEXPECTED. + + CANCEL_PUSH Frame { + Type (i) = 0x03, + Length (i), + Push ID (i), + } + + Figure 6: CANCEL_PUSH Frame + + The CANCEL_PUSH frame carries a push ID encoded as a variable-length + integer. The Push ID field identifies the server push that is being + cancelled; see Section 4.6. If a CANCEL_PUSH frame is received that + references a push ID greater than currently allowed on the + connection, this MUST be treated as a connection error of type + H3_ID_ERROR. + + If the client receives a CANCEL_PUSH frame, that frame might identify + a push ID that has not yet been mentioned by a PUSH_PROMISE frame due + to reordering. If a server receives a CANCEL_PUSH frame for a push + ID that has not yet been mentioned by a PUSH_PROMISE frame, this MUST + be treated as a connection error of type H3_ID_ERROR. + +7.2.4. SETTINGS + + The SETTINGS frame (type=0x04) conveys configuration parameters that + affect how endpoints communicate, such as preferences and constraints + on peer behavior. Individually, a SETTINGS parameter can also be + referred to as a "setting"; the identifier and value of each setting + parameter can be referred to as a "setting identifier" and a "setting + value". + + SETTINGS frames always apply to an entire HTTP/3 connection, never a + single stream. A SETTINGS frame MUST be sent as the first frame of + each control stream (see Section 6.2.1) by each peer, and it MUST NOT + be sent subsequently. If an endpoint receives a second SETTINGS + frame on the control stream, the endpoint MUST respond with a + connection error of type H3_FRAME_UNEXPECTED. + + SETTINGS frames MUST NOT be sent on any stream other than the control + stream. If an endpoint receives a SETTINGS frame on a different + stream, the endpoint MUST respond with a connection error of type + H3_FRAME_UNEXPECTED. + + SETTINGS parameters are not negotiated; they describe characteristics + of the sending peer that can be used by the receiving peer. However, + a negotiation can be implied by the use of SETTINGS: each peer uses + SETTINGS to advertise a set of supported values. The definition of + the setting would describe how each peer combines the two sets to + conclude which choice will be used. SETTINGS does not provide a + mechanism to identify when the choice takes effect. + + Different values for the same parameter can be advertised by each + peer. For example, a client might be willing to consume a very large + response field section, while servers are more cautious about request + size. + + The same setting identifier MUST NOT occur more than once in the + SETTINGS frame. A receiver MAY treat the presence of duplicate + setting identifiers as a connection error of type H3_SETTINGS_ERROR. + + The payload of a SETTINGS frame consists of zero or more parameters. + Each parameter consists of a setting identifier and a value, both + encoded as QUIC variable-length integers. + + Setting { + Identifier (i), + Value (i), + } + + SETTINGS Frame { + Type (i) = 0x04, + Length (i), + Setting (..) ..., + } + + Figure 7: SETTINGS Frame + + An implementation MUST ignore any parameter with an identifier it + does not understand. + +7.2.4.1. Defined SETTINGS Parameters + + The following settings are defined in HTTP/3: + + SETTINGS_MAX_FIELD_SECTION_SIZE (0x06): The default value is + unlimited. See Section 4.2.2 for usage. + + Setting identifiers of the format 0x1f * N + 0x21 for non-negative + integer values of N are reserved to exercise the requirement that + unknown identifiers be ignored. Such settings have no defined + meaning. Endpoints SHOULD include at least one such setting in their + SETTINGS frame. Endpoints MUST NOT consider such settings to have + any meaning upon receipt. + + Because the setting has no defined meaning, the value of the setting + can be any value the implementation selects. + + Setting identifiers that were defined in [HTTP/2] where there is no + corresponding HTTP/3 setting have also been reserved + (Section 11.2.2). These reserved settings MUST NOT be sent, and + their receipt MUST be treated as a connection error of type + H3_SETTINGS_ERROR. + + Additional settings can be defined by extensions to HTTP/3; see + Section 9 for more details. + +7.2.4.2. Initialization + + An HTTP implementation MUST NOT send frames or requests that would be + invalid based on its current understanding of the peer's settings. + + All settings begin at an initial value. Each endpoint SHOULD use + these initial values to send messages before the peer's SETTINGS + frame has arrived, as packets carrying the settings can be lost or + delayed. When the SETTINGS frame arrives, any settings are changed + to their new values. + + This removes the need to wait for the SETTINGS frame before sending + messages. Endpoints MUST NOT require any data to be received from + the peer prior to sending the SETTINGS frame; settings MUST be sent + as soon as the transport is ready to send data. + + For servers, the initial value of each client setting is the default + value. + + For clients using a 1-RTT QUIC connection, the initial value of each + server setting is the default value. 1-RTT keys will always become + available prior to the packet containing SETTINGS being processed by + QUIC, even if the server sends SETTINGS immediately. Clients SHOULD + NOT wait indefinitely for SETTINGS to arrive before sending requests, + but they SHOULD process received datagrams in order to increase the + likelihood of processing SETTINGS before sending the first request. + + When a 0-RTT QUIC connection is being used, the initial value of each + server setting is the value used in the previous session. Clients + SHOULD store the settings the server provided in the HTTP/3 + connection where resumption information was provided, but they MAY + opt not to store settings in certain cases (e.g., if the session + ticket is received before the SETTINGS frame). A client MUST comply + with stored settings -- or default values if no values are stored -- + when attempting 0-RTT. Once a server has provided new settings, + clients MUST comply with those values. + + A server can remember the settings that it advertised or store an + integrity-protected copy of the values in the ticket and recover the + information when accepting 0-RTT data. A server uses the HTTP/3 + settings values in determining whether to accept 0-RTT data. If the + server cannot determine that the settings remembered by a client are + compatible with its current settings, it MUST NOT accept 0-RTT data. + Remembered settings are compatible if a client complying with those + settings would not violate the server's current settings. + + A server MAY accept 0-RTT and subsequently provide different settings + in its SETTINGS frame. If 0-RTT data is accepted by the server, its + SETTINGS frame MUST NOT reduce any limits or alter any values that + might be violated by the client with its 0-RTT data. The server MUST + include all settings that differ from their default values. If a + server accepts 0-RTT but then sends settings that are not compatible + with the previously specified settings, this MUST be treated as a + connection error of type H3_SETTINGS_ERROR. If a server accepts + 0-RTT but then sends a SETTINGS frame that omits a setting value that + the client understands (apart from reserved setting identifiers) that + was previously specified to have a non-default value, this MUST be + treated as a connection error of type H3_SETTINGS_ERROR. + +7.2.5. PUSH_PROMISE + + The PUSH_PROMISE frame (type=0x05) is used to carry a promised + request header section from server to client on a request stream. + + PUSH_PROMISE Frame { + Type (i) = 0x05, + Length (i), + Push ID (i), + Encoded Field Section (..), + } + + Figure 8: PUSH_PROMISE Frame + + The payload consists of: + + Push ID: A variable-length integer that identifies the server push + operation. A push ID is used in push stream headers (Section 4.6) + and CANCEL_PUSH frames. + + Encoded Field Section: QPACK-encoded request header fields for the + promised response. See [QPACK] for more details. + + A server MUST NOT use a push ID that is larger than the client has + provided in a MAX_PUSH_ID frame (Section 7.2.7). A client MUST treat + receipt of a PUSH_PROMISE frame that contains a larger push ID than + the client has advertised as a connection error of H3_ID_ERROR. + + A server MAY use the same push ID in multiple PUSH_PROMISE frames. + If so, the decompressed request header sets MUST contain the same + fields in the same order, and both the name and the value in each + field MUST be exact matches. Clients SHOULD compare the request + header sections for resources promised multiple times. If a client + receives a push ID that has already been promised and detects a + mismatch, it MUST respond with a connection error of type + H3_GENERAL_PROTOCOL_ERROR. If the decompressed field sections match + exactly, the client SHOULD associate the pushed content with each + stream on which a PUSH_PROMISE frame was received. + + Allowing duplicate references to the same push ID is primarily to + reduce duplication caused by concurrent requests. A server SHOULD + avoid reusing a push ID over a long period. Clients are likely to + consume server push responses and not retain them for reuse over + time. Clients that see a PUSH_PROMISE frame that uses a push ID that + they have already consumed and discarded are forced to ignore the + promise. + + If a PUSH_PROMISE frame is received on the control stream, the client + MUST respond with a connection error of type H3_FRAME_UNEXPECTED. + + A client MUST NOT send a PUSH_PROMISE frame. A server MUST treat the + receipt of a PUSH_PROMISE frame as a connection error of type + H3_FRAME_UNEXPECTED. + + See Section 4.6 for a description of the overall server push + mechanism. + +7.2.6. GOAWAY + + The GOAWAY frame (type=0x07) is used to initiate graceful shutdown of + an HTTP/3 connection by either endpoint. GOAWAY allows an endpoint + to stop accepting new requests or pushes while still finishing + processing of previously received requests and pushes. This enables + administrative actions, like server maintenance. GOAWAY by itself + does not close a connection. + + GOAWAY Frame { + Type (i) = 0x07, + Length (i), + Stream ID/Push ID (i), + } + + Figure 9: GOAWAY Frame + + The GOAWAY frame is always sent on the control stream. In the + server-to-client direction, it carries a QUIC stream ID for a client- + initiated bidirectional stream encoded as a variable-length integer. + A client MUST treat receipt of a GOAWAY frame containing a stream ID + of any other type as a connection error of type H3_ID_ERROR. + + In the client-to-server direction, the GOAWAY frame carries a push ID + encoded as a variable-length integer. + + The GOAWAY frame applies to the entire connection, not a specific + stream. A client MUST treat a GOAWAY frame on a stream other than + the control stream as a connection error of type H3_FRAME_UNEXPECTED. + + See Section 5.2 for more information on the use of the GOAWAY frame. + +7.2.7. MAX_PUSH_ID + + The MAX_PUSH_ID frame (type=0x0d) is used by clients to control the + number of server pushes that the server can initiate. This sets the + maximum value for a push ID that the server can use in PUSH_PROMISE + and CANCEL_PUSH frames. Consequently, this also limits the number of + push streams that the server can initiate in addition to the limit + maintained by the QUIC transport. + + The MAX_PUSH_ID frame is always sent on the control stream. Receipt + of a MAX_PUSH_ID frame on any other stream MUST be treated as a + connection error of type H3_FRAME_UNEXPECTED. + + A server MUST NOT send a MAX_PUSH_ID frame. A client MUST treat the + receipt of a MAX_PUSH_ID frame as a connection error of type + H3_FRAME_UNEXPECTED. + + The maximum push ID is unset when an HTTP/3 connection is created, + meaning that a server cannot push until it receives a MAX_PUSH_ID + frame. A client that wishes to manage the number of promised server + pushes can increase the maximum push ID by sending MAX_PUSH_ID frames + as the server fulfills or cancels server pushes. + + MAX_PUSH_ID Frame { + Type (i) = 0x0d, + Length (i), + Push ID (i), + } + + Figure 10: MAX_PUSH_ID Frame + + The MAX_PUSH_ID frame carries a single variable-length integer that + identifies the maximum value for a push ID that the server can use; + see Section 4.6. A MAX_PUSH_ID frame cannot reduce the maximum push + ID; receipt of a MAX_PUSH_ID frame that contains a smaller value than + previously received MUST be treated as a connection error of type + H3_ID_ERROR. + +7.2.8. Reserved Frame Types + + Frame types of the format 0x1f * N + 0x21 for non-negative integer + values of N are reserved to exercise the requirement that unknown + types be ignored (Section 9). These frames have no semantics, and + they MAY be sent on any stream where frames are allowed to be sent. + This enables their use for application-layer padding. Endpoints MUST + NOT consider these frames to have any meaning upon receipt. + + The payload and length of the frames are selected in any manner the + implementation chooses. + + Frame types that were used in HTTP/2 where there is no corresponding + HTTP/3 frame have also been reserved (Section 11.2.1). These frame + types MUST NOT be sent, and their receipt MUST be treated as a + connection error of type H3_FRAME_UNEXPECTED. + +8. Error Handling + + When a stream cannot be completed successfully, QUIC allows the + application to abruptly terminate (reset) that stream and communicate + a reason; see Section 2.4 of [QUIC-TRANSPORT]. This is referred to + as a "stream error". An HTTP/3 implementation can decide to close a + QUIC stream and communicate the type of error. Wire encodings of + error codes are defined in Section 8.1. Stream errors are distinct + from HTTP status codes that indicate error conditions. Stream errors + indicate that the sender did not transfer or consume the full request + or response, while HTTP status codes indicate the result of a request + that was successfully received. + + If an entire connection needs to be terminated, QUIC similarly + provides mechanisms to communicate a reason; see Section 5.3 of + [QUIC-TRANSPORT]. This is referred to as a "connection error". + Similar to stream errors, an HTTP/3 implementation can terminate a + QUIC connection and communicate the reason using an error code from + Section 8.1. + + Although the reasons for closing streams and connections are called + "errors", these actions do not necessarily indicate a problem with + the connection or either implementation. For example, a stream can + be reset if the requested resource is no longer needed. + + An endpoint MAY choose to treat a stream error as a connection error + under certain circumstances, closing the entire connection in + response to a condition on a single stream. Implementations need to + consider the impact on outstanding requests before making this + choice. + + Because new error codes can be defined without negotiation (see + Section 9), use of an error code in an unexpected context or receipt + of an unknown error code MUST be treated as equivalent to + H3_NO_ERROR. However, closing a stream can have other effects + regardless of the error code; for example, see Section 4.1. + +8.1. HTTP/3 Error Codes + + The following error codes are defined for use when abruptly + terminating streams, aborting reading of streams, or immediately + closing HTTP/3 connections. + + H3_NO_ERROR (0x0100): No error. This is used when the connection or + stream needs to be closed, but there is no error to signal. + + H3_GENERAL_PROTOCOL_ERROR (0x0101): Peer violated protocol + requirements in a way that does not match a more specific error + code or endpoint declines to use the more specific error code. + + H3_INTERNAL_ERROR (0x0102): An internal error has occurred in the + HTTP stack. + + H3_STREAM_CREATION_ERROR (0x0103): The endpoint detected that its + peer created a stream that it will not accept. + + H3_CLOSED_CRITICAL_STREAM (0x0104): A stream required by the HTTP/3 + connection was closed or reset. + + H3_FRAME_UNEXPECTED (0x0105): A frame was received that was not + permitted in the current state or on the current stream. + + H3_FRAME_ERROR (0x0106): A frame that fails to satisfy layout + requirements or with an invalid size was received. + + H3_EXCESSIVE_LOAD (0x0107): The endpoint detected that its peer is + exhibiting a behavior that might be generating excessive load. + + H3_ID_ERROR (0x0108): A stream ID or push ID was used incorrectly, + such as exceeding a limit, reducing a limit, or being reused. + + H3_SETTINGS_ERROR (0x0109): An endpoint detected an error in the + payload of a SETTINGS frame. + + H3_MISSING_SETTINGS (0x010a): No SETTINGS frame was received at the + beginning of the control stream. + + H3_REQUEST_REJECTED (0x010b): A server rejected a request without + performing any application processing. + + H3_REQUEST_CANCELLED (0x010c): The request or its response + (including pushed response) is cancelled. + + H3_REQUEST_INCOMPLETE (0x010d): The client's stream terminated + without containing a fully formed request. + + H3_MESSAGE_ERROR (0x010e): An HTTP message was malformed and cannot + be processed. + + H3_CONNECT_ERROR (0x010f): The TCP connection established in + response to a CONNECT request was reset or abnormally closed. + + H3_VERSION_FALLBACK (0x0110): The requested operation cannot be + served over HTTP/3. The peer should retry over HTTP/1.1. + + Error codes of the format 0x1f * N + 0x21 for non-negative integer + values of N are reserved to exercise the requirement that unknown + error codes be treated as equivalent to H3_NO_ERROR (Section 9). + Implementations SHOULD select an error code from this space with some + probability when they would have sent H3_NO_ERROR. + +9. Extensions to HTTP/3 + + HTTP/3 permits extension of the protocol. Within the limitations + described in this section, protocol extensions can be used to provide + additional services or alter any aspect of the protocol. Extensions + are effective only within the scope of a single HTTP/3 connection. + + This applies to the protocol elements defined in this document. This + does not affect the existing options for extending HTTP, such as + defining new methods, status codes, or fields. + + Extensions are permitted to use new frame types (Section 7.2), new + settings (Section 7.2.4.1), new error codes (Section 8), or new + unidirectional stream types (Section 6.2). Registries are + established for managing these extension points: frame types + (Section 11.2.1), settings (Section 11.2.2), error codes + (Section 11.2.3), and stream types (Section 11.2.4). + + Implementations MUST ignore unknown or unsupported values in all + extensible protocol elements. Implementations MUST discard data or + abort reading on unidirectional streams that have unknown or + unsupported types. This means that any of these extension points can + be safely used by extensions without prior arrangement or + negotiation. However, where a known frame type is required to be in + a specific location, such as the SETTINGS frame as the first frame of + the control stream (see Section 6.2.1), an unknown frame type does + not satisfy that requirement and SHOULD be treated as an error. + + Extensions that could change the semantics of existing protocol + components MUST be negotiated before being used. For example, an + extension that changes the layout of the HEADERS frame cannot be used + until the peer has given a positive signal that this is acceptable. + Coordinating when such a revised layout comes into effect could prove + complex. As such, allocating new identifiers for new definitions of + existing protocol elements is likely to be more effective. + + This document does not mandate a specific method for negotiating the + use of an extension, but it notes that a setting (Section 7.2.4.1) + could be used for that purpose. If both peers set a value that + indicates willingness to use the extension, then the extension can be + used. If a setting is used for extension negotiation, the default + value MUST be defined in such a fashion that the extension is + disabled if the setting is omitted. + +10. Security Considerations + + The security considerations of HTTP/3 should be comparable to those + of HTTP/2 with TLS. However, many of the considerations from + Section 10 of [HTTP/2] apply to [QUIC-TRANSPORT] and are discussed in + that document. + +10.1. Server Authority + + HTTP/3 relies on the HTTP definition of authority. The security + considerations of establishing authority are discussed in + Section 17.1 of [HTTP]. + +10.2. Cross-Protocol Attacks + + The use of ALPN in the TLS and QUIC handshakes establishes the target + application protocol before application-layer bytes are processed. + This ensures that endpoints have strong assurances that peers are + using the same protocol. + + This does not guarantee protection from all cross-protocol attacks. + Section 21.5 of [QUIC-TRANSPORT] describes some ways in which the + plaintext of QUIC packets can be used to perform request forgery + against endpoints that don't use authenticated transports. + +10.3. Intermediary-Encapsulation Attacks + + The HTTP/3 field encoding allows the expression of names that are not + valid field names in the syntax used by HTTP (Section 5.1 of [HTTP]). + Requests or responses containing invalid field names MUST be treated + as malformed. Therefore, an intermediary cannot translate an HTTP/3 + request or response containing an invalid field name into an HTTP/1.1 + message. + + Similarly, HTTP/3 can transport field values that are not valid. + While most values that can be encoded will not alter field parsing, + carriage return (ASCII 0x0d), line feed (ASCII 0x0a), and the null + character (ASCII 0x00) might be exploited by an attacker if they are + translated verbatim. Any request or response that contains a + character not permitted in a field value MUST be treated as + malformed. Valid characters are defined by the "field-content" ABNF + rule in Section 5.5 of [HTTP]. + +10.4. Cacheability of Pushed Responses + + Pushed responses do not have an explicit request from the client; the + request is provided by the server in the PUSH_PROMISE frame. + + Caching responses that are pushed is possible based on the guidance + provided by the origin server in the Cache-Control header field. + However, this can cause issues if a single server hosts more than one + tenant. For example, a server might offer multiple users each a + small portion of its URI space. + + Where multiple tenants share space on the same server, that server + MUST ensure that tenants are not able to push representations of + resources that they do not have authority over. Failure to enforce + this would allow a tenant to provide a representation that would be + served out of cache, overriding the actual representation that the + authoritative tenant provides. + + Clients are required to reject pushed responses for which an origin + server is not authoritative; see Section 4.6. + +10.5. Denial-of-Service Considerations + + An HTTP/3 connection can demand a greater commitment of resources to + operate than an HTTP/1.1 or HTTP/2 connection. The use of field + compression and flow control depend on a commitment of resources for + storing a greater amount of state. Settings for these features + ensure that memory commitments for these features are strictly + bounded. + + The number of PUSH_PROMISE frames is constrained in a similar + fashion. A client that accepts server push SHOULD limit the number + of push IDs it issues at a time. + + Processing capacity cannot be guarded as effectively as state + capacity. + + The ability to send undefined protocol elements that the peer is + required to ignore can be abused to cause a peer to expend additional + processing time. This might be done by setting multiple undefined + SETTINGS parameters, unknown frame types, or unknown stream types. + Note, however, that some uses are entirely legitimate, such as + optional-to-understand extensions and padding to increase resistance + to traffic analysis. + + Compression of field sections also offers some opportunities to waste + processing resources; see Section 7 of [QPACK] for more details on + potential abuses. + + All these features -- i.e., server push, unknown protocol elements, + field compression -- have legitimate uses. These features become a + burden only when they are used unnecessarily or to excess. + + An endpoint that does not monitor such behavior exposes itself to a + risk of denial-of-service attack. Implementations SHOULD track the + use of these features and set limits on their use. An endpoint MAY + treat activity that is suspicious as a connection error of type + H3_EXCESSIVE_LOAD, but false positives will result in disrupting + valid connections and requests. + +10.5.1. Limits on Field Section Size + + A large field section (Section 4.1) can cause an implementation to + commit a large amount of state. Header fields that are critical for + routing can appear toward the end of a header section, which prevents + streaming of the header section to its ultimate destination. This + ordering and other reasons, such as ensuring cache correctness, mean + that an endpoint likely needs to buffer the entire header section. + Since there is no hard limit to the size of a field section, some + endpoints could be forced to commit a large amount of available + memory for header fields. + + An endpoint can use the SETTINGS_MAX_FIELD_SECTION_SIZE + (Section 4.2.2) setting to advise peers of limits that might apply on + the size of field sections. This setting is only advisory, so + endpoints MAY choose to send field sections that exceed this limit + and risk having the request or response being treated as malformed. + This setting is specific to an HTTP/3 connection, so any request or + response could encounter a hop with a lower, unknown limit. An + intermediary can attempt to avoid this problem by passing on values + presented by different peers, but they are not obligated to do so. + + A server that receives a larger field section than it is willing to + handle can send an HTTP 431 (Request Header Fields Too Large) status + code ([RFC6585]). A client can discard responses that it cannot + process. + +10.5.2. CONNECT Issues + + The CONNECT method can be used to create disproportionate load on a + proxy, since stream creation is relatively inexpensive when compared + to the creation and maintenance of a TCP connection. Therefore, a + proxy that supports CONNECT might be more conservative in the number + of simultaneous requests it accepts. + + A proxy might also maintain some resources for a TCP connection + beyond the closing of the stream that carries the CONNECT request, + since the outgoing TCP connection remains in the TIME_WAIT state. To + account for this, a proxy might delay increasing the QUIC stream + limits for some time after a TCP connection terminates. + +10.6. Use of Compression + + Compression can allow an attacker to recover secret data when it is + compressed in the same context as data under attacker control. + HTTP/3 enables compression of fields (Section 4.2); the following + concerns also apply to the use of HTTP compressed content-codings; + see Section 8.4.1 of [HTTP]. + + There are demonstrable attacks on compression that exploit the + characteristics of the web (e.g., [BREACH]). The attacker induces + multiple requests containing varying plaintext, observing the length + of the resulting ciphertext in each, which reveals a shorter length + when a guess about the secret is correct. + + Implementations communicating on a secure channel MUST NOT compress + content that includes both confidential and attacker-controlled data + unless separate compression contexts are used for each source of + data. Compression MUST NOT be used if the source of data cannot be + reliably determined. + + Further considerations regarding the compression of field sections + are described in [QPACK]. + +10.7. Padding and Traffic Analysis + + Padding can be used to obscure the exact size of frame content and is + provided to mitigate specific attacks within HTTP, for example, + attacks where compressed content includes both attacker-controlled + plaintext and secret data (e.g., [BREACH]). + + Where HTTP/2 employs PADDING frames and Padding fields in other + frames to make a connection more resistant to traffic analysis, + HTTP/3 can either rely on transport-layer padding or employ the + reserved frame and stream types discussed in Sections 7.2.8 and + 6.2.3. These methods of padding produce different results in terms + of the granularity of padding, how padding is arranged in relation to + the information that is being protected, whether padding is applied + in the case of packet loss, and how an implementation might control + padding. + + Reserved stream types can be used to give the appearance of sending + traffic even when the connection is idle. Because HTTP traffic often + occurs in bursts, apparent traffic can be used to obscure the timing + or duration of such bursts, even to the point of appearing to send a + constant stream of data. However, as such traffic is still flow + controlled by the receiver, a failure to promptly drain such streams + and provide additional flow-control credit can limit the sender's + ability to send real traffic. + + To mitigate attacks that rely on compression, disabling or limiting + compression might be preferable to padding as a countermeasure. + + Use of padding can result in less protection than might seem + immediately obvious. Redundant padding could even be + counterproductive. At best, padding only makes it more difficult for + an attacker to infer length information by increasing the number of + frames an attacker has to observe. Incorrectly implemented padding + schemes can be easily defeated. In particular, randomized padding + with a predictable distribution provides very little protection; + similarly, padding payloads to a fixed size exposes information as + payload sizes cross the fixed-sized boundary, which could be possible + if an attacker can control plaintext. + +10.8. Frame Parsing + + Several protocol elements contain nested length elements, typically + in the form of frames with an explicit length containing variable- + length integers. This could pose a security risk to an incautious + implementer. An implementation MUST ensure that the length of a + frame exactly matches the length of the fields it contains. + +10.9. Early Data + + The use of 0-RTT with HTTP/3 creates an exposure to replay attack. + The anti-replay mitigations in [HTTP-REPLAY] MUST be applied when + using HTTP/3 with 0-RTT. When applying [HTTP-REPLAY] to HTTP/3, + references to the TLS layer refer to the handshake performed within + QUIC, while all references to application data refer to the contents + of streams. + +10.10. Migration + + Certain HTTP implementations use the client address for logging or + access-control purposes. Since a QUIC client's address might change + during a connection (and future versions might support simultaneous + use of multiple addresses), such implementations will need to either + actively retrieve the client's current address or addresses when they + are relevant or explicitly accept that the original address might + change. + +10.11. Privacy Considerations + + Several characteristics of HTTP/3 provide an observer an opportunity + to correlate actions of a single client or server over time. These + include the value of settings, the timing of reactions to stimulus, + and the handling of any features that are controlled by settings. + + As far as these create observable differences in behavior, they could + be used as a basis for fingerprinting a specific client. + + HTTP/3's preference for using a single QUIC connection allows + correlation of a user's activity on a site. Reusing connections for + different origins allows for correlation of activity across those + origins. + + Several features of QUIC solicit immediate responses and can be used + by an endpoint to measure latency to their peer; this might have + privacy implications in certain scenarios. + +11. IANA Considerations + + This document registers a new ALPN protocol ID (Section 11.1) and + creates new registries that manage the assignment of code points in + HTTP/3. + +11.1. Registration of HTTP/3 Identification String + + This document creates a new registration for the identification of + HTTP/3 in the "TLS Application-Layer Protocol Negotiation (ALPN) + Protocol IDs" registry established in [RFC7301]. + + The "h3" string identifies HTTP/3: + + Protocol: HTTP/3 + + Identification Sequence: 0x68 0x33 ("h3") + + Specification: This document + +11.2. New Registries + + New registries created in this document operate under the QUIC + registration policy documented in Section 22.1 of [QUIC-TRANSPORT]. + These registries all include the common set of fields listed in + Section 22.1.1 of [QUIC-TRANSPORT]. These registries are collected + under the "Hypertext Transfer Protocol version 3 (HTTP/3)" heading. + + The initial allocations in these registries are all assigned + permanent status and list a change controller of the IETF and a + contact of the HTTP working group (ietf-http-wg@w3.org). + +11.2.1. Frame Types + + This document establishes a registry for HTTP/3 frame type codes. + The "HTTP/3 Frame Types" registry governs a 62-bit space. This + registry follows the QUIC registry policy; see Section 11.2. + Permanent registrations in this registry are assigned using the + Specification Required policy ([RFC8126]), except for values between + 0x00 and 0x3f (in hexadecimal; inclusive), which are assigned using + Standards Action or IESG Approval as defined in Sections 4.9 and 4.10 + of [RFC8126]. + + While this registry is separate from the "HTTP/2 Frame Type" registry + defined in [HTTP/2], it is preferable that the assignments parallel + each other where the code spaces overlap. If an entry is present in + only one registry, every effort SHOULD be made to avoid assigning the + corresponding value to an unrelated operation. Expert reviewers MAY + reject unrelated registrations that would conflict with the same + value in the corresponding registry. + + In addition to common fields as described in Section 11.2, permanent + registrations in this registry MUST include the following field: + + Frame Type: A name or label for the frame type. + + Specifications of frame types MUST include a description of the frame + layout and its semantics, including any parts of the frame that are + conditionally present. + + The entries in Table 2 are registered by this document. + + +==============+=======+===============+ + | Frame Type | Value | Specification | + +==============+=======+===============+ + | DATA | 0x00 | Section 7.2.1 | + +--------------+-------+---------------+ + | HEADERS | 0x01 | Section 7.2.2 | + +--------------+-------+---------------+ + | Reserved | 0x02 | This document | + +--------------+-------+---------------+ + | CANCEL_PUSH | 0x03 | Section 7.2.3 | + +--------------+-------+---------------+ + | SETTINGS | 0x04 | Section 7.2.4 | + +--------------+-------+---------------+ + | PUSH_PROMISE | 0x05 | Section 7.2.5 | + +--------------+-------+---------------+ + | Reserved | 0x06 | This document | + +--------------+-------+---------------+ + | GOAWAY | 0x07 | Section 7.2.6 | + +--------------+-------+---------------+ + | Reserved | 0x08 | This document | + +--------------+-------+---------------+ + | Reserved | 0x09 | This document | + +--------------+-------+---------------+ + | MAX_PUSH_ID | 0x0d | Section 7.2.7 | + +--------------+-------+---------------+ + + Table 2: Initial HTTP/3 Frame Types + + Each code of the format 0x1f * N + 0x21 for non-negative integer + values of N (that is, 0x21, 0x40, ..., through 0x3ffffffffffffffe) + MUST NOT be assigned by IANA and MUST NOT appear in the listing of + assigned values. + +11.2.2. Settings Parameters + + This document establishes a registry for HTTP/3 settings. The + "HTTP/3 Settings" registry governs a 62-bit space. This registry + follows the QUIC registry policy; see Section 11.2. Permanent + registrations in this registry are assigned using the Specification + Required policy ([RFC8126]), except for values between 0x00 and 0x3f + (in hexadecimal; inclusive), which are assigned using Standards + Action or IESG Approval as defined in Sections 4.9 and 4.10 of + [RFC8126]. + + While this registry is separate from the "HTTP/2 Settings" registry + defined in [HTTP/2], it is preferable that the assignments parallel + each other. If an entry is present in only one registry, every + effort SHOULD be made to avoid assigning the corresponding value to + an unrelated operation. Expert reviewers MAY reject unrelated + registrations that would conflict with the same value in the + corresponding registry. + + In addition to common fields as described in Section 11.2, permanent + registrations in this registry MUST include the following fields: + + Setting Name: A symbolic name for the setting. Specifying a setting + name is optional. + + Default: The value of the setting unless otherwise indicated. A + default SHOULD be the most restrictive possible value. + + The entries in Table 3 are registered by this document. + + +========================+=======+=================+===========+ + | Setting Name | Value | Specification | Default | + +========================+=======+=================+===========+ + | Reserved | 0x00 | This document | N/A | + +------------------------+-------+-----------------+-----------+ + | Reserved | 0x02 | This document | N/A | + +------------------------+-------+-----------------+-----------+ + | Reserved | 0x03 | This document | N/A | + +------------------------+-------+-----------------+-----------+ + | Reserved | 0x04 | This document | N/A | + +------------------------+-------+-----------------+-----------+ + | Reserved | 0x05 | This document | N/A | + +------------------------+-------+-----------------+-----------+ + | MAX_FIELD_SECTION_SIZE | 0x06 | Section 7.2.4.1 | Unlimited | + +------------------------+-------+-----------------+-----------+ + + Table 3: Initial HTTP/3 Settings + + For formatting reasons, setting names can be abbreviated by removing + the 'SETTINGS_' prefix. + + Each code of the format 0x1f * N + 0x21 for non-negative integer + values of N (that is, 0x21, 0x40, ..., through 0x3ffffffffffffffe) + MUST NOT be assigned by IANA and MUST NOT appear in the listing of + assigned values. + +11.2.3. Error Codes + + This document establishes a registry for HTTP/3 error codes. The + "HTTP/3 Error Codes" registry manages a 62-bit space. This registry + follows the QUIC registry policy; see Section 11.2. Permanent + registrations in this registry are assigned using the Specification + Required policy ([RFC8126]), except for values between 0x00 and 0x3f + (in hexadecimal; inclusive), which are assigned using Standards + Action or IESG Approval as defined in Sections 4.9 and 4.10 of + [RFC8126]. + + Registrations for error codes are required to include a description + of the error code. An expert reviewer is advised to examine new + registrations for possible duplication with existing error codes. + Use of existing registrations is to be encouraged, but not mandated. + Use of values that are registered in the "HTTP/2 Error Code" registry + is discouraged, and expert reviewers MAY reject such registrations. + + In addition to common fields as described in Section 11.2, this + registry includes two additional fields. Permanent registrations in + this registry MUST include the following field: + + Name: A name for the error code. + + Description: A brief description of the error code semantics. + + The entries in Table 4 are registered by this document. These error + codes were selected from the range that operates on a Specification + Required policy to avoid collisions with HTTP/2 error codes. + + +===========================+========+==============+===============+ + | Name | Value | Description | Specification | + +===========================+========+==============+===============+ + | H3_NO_ERROR | 0x0100 | No error | Section 8.1 | + +---------------------------+--------+--------------+---------------+ + | H3_GENERAL_PROTOCOL_ERROR | 0x0101 | General | Section 8.1 | + | | | protocol | | + | | | error | | + +---------------------------+--------+--------------+---------------+ + | H3_INTERNAL_ERROR | 0x0102 | Internal | Section 8.1 | + | | | error | | + +---------------------------+--------+--------------+---------------+ + | H3_STREAM_CREATION_ERROR | 0x0103 | Stream | Section 8.1 | + | | | creation | | + | | | error | | + +---------------------------+--------+--------------+---------------+ + | H3_CLOSED_CRITICAL_STREAM | 0x0104 | Critical | Section 8.1 | + | | | stream was | | + | | | closed | | + +---------------------------+--------+--------------+---------------+ + | H3_FRAME_UNEXPECTED | 0x0105 | Frame not | Section 8.1 | + | | | permitted | | + | | | in the | | + | | | current | | + | | | state | | + +---------------------------+--------+--------------+---------------+ + | H3_FRAME_ERROR | 0x0106 | Frame | Section 8.1 | + | | | violated | | + | | | layout or | | + | | | size rules | | + +---------------------------+--------+--------------+---------------+ + | H3_EXCESSIVE_LOAD | 0x0107 | Peer | Section 8.1 | + | | | generating | | + | | | excessive | | + | | | load | | + +---------------------------+--------+--------------+---------------+ + | H3_ID_ERROR | 0x0108 | An | Section 8.1 | + | | | identifier | | + | | | was used | | + | | | incorrectly | | + +---------------------------+--------+--------------+---------------+ + | H3_SETTINGS_ERROR | 0x0109 | SETTINGS | Section 8.1 | + | | | frame | | + | | | contained | | + | | | invalid | | + | | | values | | + +---------------------------+--------+--------------+---------------+ + | H3_MISSING_SETTINGS | 0x010a | No SETTINGS | Section 8.1 | + | | | frame | | + | | | received | | + +---------------------------+--------+--------------+---------------+ + | H3_REQUEST_REJECTED | 0x010b | Request not | Section 8.1 | + | | | processed | | + +---------------------------+--------+--------------+---------------+ + | H3_REQUEST_CANCELLED | 0x010c | Data no | Section 8.1 | + | | | longer | | + | | | needed | | + +---------------------------+--------+--------------+---------------+ + | H3_REQUEST_INCOMPLETE | 0x010d | Stream | Section 8.1 | + | | | terminated | | + | | | early | | + +---------------------------+--------+--------------+---------------+ + | H3_MESSAGE_ERROR | 0x010e | Malformed | Section 8.1 | + | | | message | | + +---------------------------+--------+--------------+---------------+ + | H3_CONNECT_ERROR | 0x010f | TCP reset | Section 8.1 | + | | | or error on | | + | | | CONNECT | | + | | | request | | + +---------------------------+--------+--------------+---------------+ + | H3_VERSION_FALLBACK | 0x0110 | Retry over | Section 8.1 | + | | | HTTP/1.1 | | + +---------------------------+--------+--------------+---------------+ + + Table 4: Initial HTTP/3 Error Codes + + Each code of the format 0x1f * N + 0x21 for non-negative integer + values of N (that is, 0x21, 0x40, ..., through 0x3ffffffffffffffe) + MUST NOT be assigned by IANA and MUST NOT appear in the listing of + assigned values. + +11.2.4. Stream Types + + This document establishes a registry for HTTP/3 unidirectional stream + types. The "HTTP/3 Stream Types" registry governs a 62-bit space. + This registry follows the QUIC registry policy; see Section 11.2. + Permanent registrations in this registry are assigned using the + Specification Required policy ([RFC8126]), except for values between + 0x00 and 0x3f (in hexadecimal; inclusive), which are assigned using + Standards Action or IESG Approval as defined in Sections 4.9 and 4.10 + of [RFC8126]. + + In addition to common fields as described in Section 11.2, permanent + registrations in this registry MUST include the following fields: + + Stream Type: A name or label for the stream type. + + Sender: Which endpoint on an HTTP/3 connection may initiate a stream + of this type. Values are "Client", "Server", or "Both". + + Specifications for permanent registrations MUST include a description + of the stream type, including the layout and semantics of the stream + contents. + + The entries in Table 5 are registered by this document. + + +================+=======+===============+========+ + | Stream Type | Value | Specification | Sender | + +================+=======+===============+========+ + | Control Stream | 0x00 | Section 6.2.1 | Both | + +----------------+-------+---------------+--------+ + | Push Stream | 0x01 | Section 4.6 | Server | + +----------------+-------+---------------+--------+ + + Table 5: Initial Stream Types + + Each code of the format 0x1f * N + 0x21 for non-negative integer + values of N (that is, 0x21, 0x40, ..., through 0x3ffffffffffffffe) + MUST NOT be assigned by IANA and MUST NOT appear in the listing of + assigned values. + +12. References + +12.1. Normative References + + [ALTSVC] Nottingham, M., McManus, P., and J. Reschke, "HTTP + Alternative Services", RFC 7838, DOI 10.17487/RFC7838, + April 2016, <https://www.rfc-editor.org/info/rfc7838>. + + [COOKIES] Barth, A., "HTTP State Management Mechanism", RFC 6265, + DOI 10.17487/RFC6265, April 2011, + <https://www.rfc-editor.org/info/rfc6265>. + + [HTTP] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP Semantics", STD 97, RFC 9110, + DOI 10.17487/RFC9110, June 2022, + <https://www.rfc-editor.org/info/rfc9110>. + + [HTTP-CACHING] + Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP Caching", STD 98, RFC 9111, + DOI 10.17487/RFC9111, June 2022, + <https://www.rfc-editor.org/info/rfc9111>. + + [HTTP-REPLAY] + Thomson, M., Nottingham, M., and W. Tarreau, "Using Early + Data in HTTP", RFC 8470, DOI 10.17487/RFC8470, September + 2018, <https://www.rfc-editor.org/info/rfc8470>. + + [QPACK] Krasic, C., Bishop, M., and A. Frindell, Ed., "QPACK: + Field Compression for HTTP/3", RFC 9204, + DOI 10.17487/RFC9204, June 2022, + <https://www.rfc-editor.org/info/rfc9204>. + + [QUIC-TRANSPORT] + Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [RFC0793] Postel, J., "Transmission Control Protocol", STD 7, + RFC 793, DOI 10.17487/RFC0793, September 1981, + <https://www.rfc-editor.org/info/rfc793>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC6066] Eastlake 3rd, D., "Transport Layer Security (TLS) + Extensions: Extension Definitions", RFC 6066, + DOI 10.17487/RFC6066, January 2011, + <https://www.rfc-editor.org/info/rfc6066>. + + [RFC7301] Friedl, S., Popov, A., Langley, A., and E. Stephan, + "Transport Layer Security (TLS) Application-Layer Protocol + Negotiation Extension", RFC 7301, DOI 10.17487/RFC7301, + July 2014, <https://www.rfc-editor.org/info/rfc7301>. + + [RFC8126] Cotton, M., Leiba, B., and T. Narten, "Guidelines for + Writing an IANA Considerations Section in RFCs", BCP 26, + RFC 8126, DOI 10.17487/RFC8126, June 2017, + <https://www.rfc-editor.org/info/rfc8126>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + + [URI] Berners-Lee, T., Fielding, R., and L. Masinter, "Uniform + Resource Identifier (URI): Generic Syntax", STD 66, + RFC 3986, DOI 10.17487/RFC3986, January 2005, + <https://www.rfc-editor.org/info/rfc3986>. + +12.2. Informative References + + [BREACH] Gluck, Y., Harris, N., and A. Prado, "BREACH: Reviving the + CRIME Attack", July 2013, + <http://breachattack.com/resources/ + BREACH%20-%20SSL,%20gone%20in%2030%20seconds.pdf>. + + [DNS-TERMS] + Hoffman, P., Sullivan, A., and K. Fujiwara, "DNS + Terminology", BCP 219, RFC 8499, DOI 10.17487/RFC8499, + January 2019, <https://www.rfc-editor.org/info/rfc8499>. + + [HPACK] Peon, R. and H. Ruellan, "HPACK: Header Compression for + HTTP/2", RFC 7541, DOI 10.17487/RFC7541, May 2015, + <https://www.rfc-editor.org/info/rfc7541>. + + [HTTP/1.1] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP/1.1", STD 99, RFC 9112, DOI 10.17487/RFC9112, + June 2022, <https://www.rfc-editor.org/info/rfc9112>. + + [HTTP/2] Thomson, M., Ed. and C. Benfield, Ed., "HTTP/2", RFC 9113, + DOI 10.17487/RFC9113, June 2022, + <https://www.rfc-editor.org/info/rfc9113>. + + [RFC6585] Nottingham, M. and R. Fielding, "Additional HTTP Status + Codes", RFC 6585, DOI 10.17487/RFC6585, April 2012, + <https://www.rfc-editor.org/info/rfc6585>. + + [RFC8164] Nottingham, M. and M. Thomson, "Opportunistic Security for + HTTP/2", RFC 8164, DOI 10.17487/RFC8164, May 2017, + <https://www.rfc-editor.org/info/rfc8164>. + + [TFO] Cheng, Y., Chu, J., Radhakrishnan, S., and A. Jain, "TCP + Fast Open", RFC 7413, DOI 10.17487/RFC7413, December 2014, + <https://www.rfc-editor.org/info/rfc7413>. + + [TLS] Rescorla, E., "The Transport Layer Security (TLS) Protocol + Version 1.3", RFC 8446, DOI 10.17487/RFC8446, August 2018, + <https://www.rfc-editor.org/info/rfc8446>. + +Appendix A. Considerations for Transitioning from HTTP/2 + + HTTP/3 is strongly informed by HTTP/2, and it bears many + similarities. This section describes the approach taken to design + HTTP/3, points out important differences from HTTP/2, and describes + how to map HTTP/2 extensions into HTTP/3. + + HTTP/3 begins from the premise that similarity to HTTP/2 is + preferable, but not a hard requirement. HTTP/3 departs from HTTP/2 + where QUIC differs from TCP, either to take advantage of QUIC + features (like streams) or to accommodate important shortcomings + (such as a lack of total ordering). While HTTP/3 is similar to + HTTP/2 in key aspects, such as the relationship of requests and + responses to streams, the details of the HTTP/3 design are + substantially different from HTTP/2. + + Some important departures are noted in this section. + +A.1. Streams + + HTTP/3 permits use of a larger number of streams (2^62-1) than + HTTP/2. The same considerations about exhaustion of stream + identifier space apply, though the space is significantly larger such + that it is likely that other limits in QUIC are reached first, such + as the limit on the connection flow-control window. + + In contrast to HTTP/2, stream concurrency in HTTP/3 is managed by + QUIC. QUIC considers a stream closed when all data has been received + and sent data has been acknowledged by the peer. HTTP/2 considers a + stream closed when the frame containing the END_STREAM bit has been + committed to the transport. As a result, the stream for an + equivalent exchange could remain "active" for a longer period of + time. HTTP/3 servers might choose to permit a larger number of + concurrent client-initiated bidirectional streams to achieve + equivalent concurrency to HTTP/2, depending on the expected usage + patterns. + + In HTTP/2, only request and response bodies (the frame payload of + DATA frames) are subject to flow control. All HTTP/3 frames are sent + on QUIC streams, so all frames on all streams are flow controlled in + HTTP/3. + + Due to the presence of other unidirectional stream types, HTTP/3 does + not rely exclusively on the number of concurrent unidirectional + streams to control the number of concurrent in-flight pushes. + Instead, HTTP/3 clients use the MAX_PUSH_ID frame to control the + number of pushes received from an HTTP/3 server. + +A.2. HTTP Frame Types + + Many framing concepts from HTTP/2 can be elided on QUIC, because the + transport deals with them. Because frames are already on a stream, + they can omit the stream number. Because frames do not block + multiplexing (QUIC's multiplexing occurs below this layer), the + support for variable-maximum-length packets can be removed. Because + stream termination is handled by QUIC, an END_STREAM flag is not + required. This permits the removal of the Flags field from the + generic frame layout. + + Frame payloads are largely drawn from [HTTP/2]. However, QUIC + includes many features (e.g., flow control) that are also present in + HTTP/2. In these cases, the HTTP mapping does not re-implement them. + As a result, several HTTP/2 frame types are not required in HTTP/3. + Where an HTTP/2-defined frame is no longer used, the frame ID has + been reserved in order to maximize portability between HTTP/2 and + HTTP/3 implementations. However, even frame types that appear in + both mappings do not have identical semantics. + + Many of the differences arise from the fact that HTTP/2 provides an + absolute ordering between frames across all streams, while QUIC + provides this guarantee on each stream only. As a result, if a frame + type makes assumptions that frames from different streams will still + be received in the order sent, HTTP/3 will break them. + + Some examples of feature adaptations are described below, as well as + general guidance to extension frame implementors converting an HTTP/2 + extension to HTTP/3. + +A.2.1. Prioritization Differences + + HTTP/2 specifies priority assignments in PRIORITY frames and + (optionally) in HEADERS frames. HTTP/3 does not provide a means of + signaling priority. + + Note that, while there is no explicit signaling for priority, this + does not mean that prioritization is not important for achieving good + performance. + +A.2.2. Field Compression Differences + + HPACK was designed with the assumption of in-order delivery. A + sequence of encoded field sections must arrive (and be decoded) at an + endpoint in the same order in which they were encoded. This ensures + that the dynamic state at the two endpoints remains in sync. + + Because this total ordering is not provided by QUIC, HTTP/3 uses a + modified version of HPACK, called QPACK. QPACK uses a single + unidirectional stream to make all modifications to the dynamic table, + ensuring a total order of updates. All frames that contain encoded + fields merely reference the table state at a given time without + modifying it. + + [QPACK] provides additional details. + +A.2.3. Flow-Control Differences + + HTTP/2 specifies a stream flow-control mechanism. Although all + HTTP/2 frames are delivered on streams, only the DATA frame payload + is subject to flow control. QUIC provides flow control for stream + data and all HTTP/3 frame types defined in this document are sent on + streams. Therefore, all frame headers and payload are subject to + flow control. + +A.2.4. Guidance for New Frame Type Definitions + + Frame type definitions in HTTP/3 often use the QUIC variable-length + integer encoding. In particular, stream IDs use this encoding, which + allows for a larger range of possible values than the encoding used + in HTTP/2. Some frames in HTTP/3 use an identifier other than a + stream ID (e.g., push IDs). Redefinition of the encoding of + extension frame types might be necessary if the encoding includes a + stream ID. + + Because the Flags field is not present in generic HTTP/3 frames, + those frames that depend on the presence of flags need to allocate + space for flags as part of their frame payload. + + Other than these issues, frame type HTTP/2 extensions are typically + portable to QUIC simply by replacing stream 0 in HTTP/2 with a + control stream in HTTP/3. HTTP/3 extensions will not assume + ordering, but would not be harmed by ordering, and are expected to be + portable to HTTP/2. + +A.2.5. Comparison of HTTP/2 and HTTP/3 Frame Types + + DATA (0x00): Padding is not defined in HTTP/3 frames. See + Section 7.2.1. + + HEADERS (0x01): The PRIORITY region of HEADERS is not defined in + HTTP/3 frames. Padding is not defined in HTTP/3 frames. See + Section 7.2.2. + + PRIORITY (0x02): As described in Appendix A.2.1, HTTP/3 does not + provide a means of signaling priority. + + RST_STREAM (0x03): RST_STREAM frames do not exist in HTTP/3, since + QUIC provides stream lifecycle management. The same code point is + used for the CANCEL_PUSH frame (Section 7.2.3). + + SETTINGS (0x04): SETTINGS frames are sent only at the beginning of + the connection. See Section 7.2.4 and Appendix A.3. + + PUSH_PROMISE (0x05): The PUSH_PROMISE frame does not reference a + stream; instead, the push stream references the PUSH_PROMISE frame + using a push ID. See Section 7.2.5. + + PING (0x06): PING frames do not exist in HTTP/3, as QUIC provides + equivalent functionality. + + GOAWAY (0x07): GOAWAY does not contain an error code. In the + client-to-server direction, it carries a push ID instead of a + server-initiated stream ID. See Section 7.2.6. + + WINDOW_UPDATE (0x08): WINDOW_UPDATE frames do not exist in HTTP/3, + since QUIC provides flow control. + + CONTINUATION (0x09): CONTINUATION frames do not exist in HTTP/3; + instead, larger HEADERS/PUSH_PROMISE frames than HTTP/2 are + permitted. + + Frame types defined by extensions to HTTP/2 need to be separately + registered for HTTP/3 if still applicable. The IDs of frames defined + in [HTTP/2] have been reserved for simplicity. Note that the frame + type space in HTTP/3 is substantially larger (62 bits versus 8 bits), + so many HTTP/3 frame types have no equivalent HTTP/2 code points. + See Section 11.2.1. + +A.3. HTTP/2 SETTINGS Parameters + + An important difference from HTTP/2 is that settings are sent once, + as the first frame of the control stream, and thereafter cannot + change. This eliminates many corner cases around synchronization of + changes. + + Some transport-level options that HTTP/2 specifies via the SETTINGS + frame are superseded by QUIC transport parameters in HTTP/3. The + HTTP-level setting that is retained in HTTP/3 has the same value as + in HTTP/2. The superseded settings are reserved, and their receipt + is an error. See Section 7.2.4.1 for discussion of both the retained + and reserved values. + + Below is a listing of how each HTTP/2 SETTINGS parameter is mapped: + + SETTINGS_HEADER_TABLE_SIZE (0x01): See [QPACK]. + + SETTINGS_ENABLE_PUSH (0x02): This is removed in favor of the + MAX_PUSH_ID frame, which provides a more granular control over + server push. Specifying a setting with the identifier 0x02 + (corresponding to the SETTINGS_ENABLE_PUSH parameter) in the + HTTP/3 SETTINGS frame is an error. + + SETTINGS_MAX_CONCURRENT_STREAMS (0x03): QUIC controls the largest + open stream ID as part of its flow-control logic. Specifying a + setting with the identifier 0x03 (corresponding to the + SETTINGS_MAX_CONCURRENT_STREAMS parameter) in the HTTP/3 SETTINGS + frame is an error. + + SETTINGS_INITIAL_WINDOW_SIZE (0x04): QUIC requires both stream and + connection flow-control window sizes to be specified in the + initial transport handshake. Specifying a setting with the + identifier 0x04 (corresponding to the SETTINGS_INITIAL_WINDOW_SIZE + parameter) in the HTTP/3 SETTINGS frame is an error. + + SETTINGS_MAX_FRAME_SIZE (0x05): This setting has no equivalent in + HTTP/3. Specifying a setting with the identifier 0x05 + (corresponding to the SETTINGS_MAX_FRAME_SIZE parameter) in the + HTTP/3 SETTINGS frame is an error. + + SETTINGS_MAX_HEADER_LIST_SIZE (0x06): This setting identifier has + been renamed SETTINGS_MAX_FIELD_SECTION_SIZE. + + In HTTP/3, setting values are variable-length integers (6, 14, 30, or + 62 bits long) rather than fixed-length 32-bit fields as in HTTP/2. + This will often produce a shorter encoding, but can produce a longer + encoding for settings that use the full 32-bit space. Settings + ported from HTTP/2 might choose to redefine their value to limit it + to 30 bits for more efficient encoding or to make use of the 62-bit + space if more than 30 bits are required. + + Settings need to be defined separately for HTTP/2 and HTTP/3. The + IDs of settings defined in [HTTP/2] have been reserved for + simplicity. Note that the settings identifier space in HTTP/3 is + substantially larger (62 bits versus 16 bits), so many HTTP/3 + settings have no equivalent HTTP/2 code point. See Section 11.2.2. + + As QUIC streams might arrive out of order, endpoints are advised not + to wait for the peers' settings to arrive before responding to other + streams. See Section 7.2.4.2. + +A.4. HTTP/2 Error Codes + + QUIC has the same concepts of "stream" and "connection" errors that + HTTP/2 provides. However, the differences between HTTP/2 and HTTP/3 + mean that error codes are not directly portable between versions. + + The HTTP/2 error codes defined in Section 7 of [HTTP/2] logically map + to the HTTP/3 error codes as follows: + + NO_ERROR (0x00): H3_NO_ERROR in Section 8.1. + + PROTOCOL_ERROR (0x01): This is mapped to H3_GENERAL_PROTOCOL_ERROR + except in cases where more specific error codes have been defined. + Such cases include H3_FRAME_UNEXPECTED, H3_MESSAGE_ERROR, and + H3_CLOSED_CRITICAL_STREAM defined in Section 8.1. + + INTERNAL_ERROR (0x02): H3_INTERNAL_ERROR in Section 8.1. + + FLOW_CONTROL_ERROR (0x03): Not applicable, since QUIC handles flow + control. + + SETTINGS_TIMEOUT (0x04): Not applicable, since no acknowledgment of + SETTINGS is defined. + + STREAM_CLOSED (0x05): Not applicable, since QUIC handles stream + management. + + FRAME_SIZE_ERROR (0x06): H3_FRAME_ERROR error code defined in + Section 8.1. + + REFUSED_STREAM (0x07): H3_REQUEST_REJECTED (in Section 8.1) is used + to indicate that a request was not processed. Otherwise, not + applicable because QUIC handles stream management. + + CANCEL (0x08): H3_REQUEST_CANCELLED in Section 8.1. + + COMPRESSION_ERROR (0x09): Multiple error codes are defined in + [QPACK]. + + CONNECT_ERROR (0x0a): H3_CONNECT_ERROR in Section 8.1. + + ENHANCE_YOUR_CALM (0x0b): H3_EXCESSIVE_LOAD in Section 8.1. + + INADEQUATE_SECURITY (0x0c): Not applicable, since QUIC is assumed to + provide sufficient security on all connections. + + HTTP_1_1_REQUIRED (0x0d): H3_VERSION_FALLBACK in Section 8.1. + + Error codes need to be defined for HTTP/2 and HTTP/3 separately. See + Section 11.2.3. + +A.4.1. Mapping between HTTP/2 and HTTP/3 Errors + + An intermediary that converts between HTTP/2 and HTTP/3 may encounter + error conditions from either upstream. It is useful to communicate + the occurrence of errors to the downstream, but error codes largely + reflect connection-local problems that generally do not make sense to + propagate. + + An intermediary that encounters an error from an upstream origin can + indicate this by sending an HTTP status code such as 502 (Bad + Gateway), which is suitable for a broad class of errors. + + There are some rare cases where it is beneficial to propagate the + error by mapping it to the closest matching error type to the + receiver. For example, an intermediary that receives an HTTP/2 + stream error of type REFUSED_STREAM from the origin has a clear + signal that the request was not processed and that the request is + safe to retry. Propagating this error condition to the client as an + HTTP/3 stream error of type H3_REQUEST_REJECTED allows the client to + take the action it deems most appropriate. In the reverse direction, + the intermediary might deem it beneficial to pass on client request + cancellations that are indicated by terminating a stream with + H3_REQUEST_CANCELLED; see Section 4.1.1. + + Conversion between errors is described in the logical mapping. The + error codes are defined in non-overlapping spaces in order to protect + against accidental conversion that could result in the use of + inappropriate or unknown error codes for the target version. An + intermediary is permitted to promote stream errors to connection + errors but they should be aware of the cost to the HTTP/3 connection + for what might be a temporary or intermittent error. + +Acknowledgments + + Robbie Shade and Mike Warres were the authors of draft-shade-quic- + http2-mapping, a precursor of this document. + + The IETF QUIC Working Group received an enormous amount of support + from many people. Among others, the following people provided + substantial contributions to this document: + + * Bence Beky + * Daan De Meyer + * Martin Duke + * Roy Fielding + * Alan Frindell + * Alessandro Ghedini + * Nick Harper + * Ryan Hamilton + * Christian Huitema + * Subodh Iyengar + * Robin Marx + * Patrick McManus + * Luca Niccolini + * ๅฅฅ ไธ€็ฉ‚ (Kazuho Oku) + * Lucas Pardue + * Roberto Peon + * Julian Reschke + * Eric Rescorla + * Martin Seemann + * Ben Schwartz + * Ian Swett + * Willy Taureau + * Martin Thomson + * Dmitri Tikhonov + * Tatsuhiro Tsujikawa + + A portion of Mike Bishop's contribution was supported by Microsoft + during his employment there. + +Index + + C D G H M P R S + + C + + CANCEL_PUSH Section 2, Paragraph 5; Section 4.6, Paragraph 6; + Section 4.6, Paragraph 10; Table 1; *_Section 7.2.3_*; + Section 7.2.5, Paragraph 4.2.1; Section 7.2.7, Paragraph 1; + Table 2; Appendix A.2.5, Paragraph 1.8.1 + connection error Section 2.2; Section 4.1, Paragraph 7; + Section 4.1, Paragraph 8; Section 4.4, Paragraph 8; + Section 4.4, Paragraph 10; Section 4.6, Paragraph 3; + Section 5.2, Paragraph 7; Section 6.1, Paragraph 3; + Section 6.2, Paragraph 7; Section 6.2.1, Paragraph 2; + Section 6.2.1, Paragraph 2; Section 6.2.1, Paragraph 2; + Section 6.2.2, Paragraph 3; Section 6.2.2, Paragraph 6; + Section 7.1, Paragraph 5; Section 7.1, Paragraph 6; + Section 7.2.1, Paragraph 2; Section 7.2.2, Paragraph 3; + Section 7.2.3, Paragraph 5; Section 7.2.3, Paragraph 7; + Section 7.2.3, Paragraph 8; Section 7.2.4, Paragraph 2; + Section 7.2.4, Paragraph 3; Section 7.2.4, Paragraph 6; + Section 7.2.4.1, Paragraph 5; Section 7.2.4.2, Paragraph 8; + Section 7.2.4.2, Paragraph 8; Section 7.2.5, Paragraph 5; + Section 7.2.5, Paragraph 6; Section 7.2.5, Paragraph 8; + Section 7.2.5, Paragraph 9; Section 7.2.6, Paragraph 3; + Section 7.2.6, Paragraph 5; Section 7.2.7, Paragraph 2; + Section 7.2.7, Paragraph 3; Section 7.2.7, Paragraph 6; + Section 7.2.8, Paragraph 3; *_Section 8_*; Section 10.5, + Paragraph 7; Appendix A.4.1, Paragraph 4 + control stream Section 2, Paragraph 3; Section 3.2, Paragraph + 4; Section 6.2, Paragraph 3; Section 6.2, Paragraph 5; + Section 6.2, Paragraph 6; *_Section 6.2.1_*; Section 7, + Paragraph 1; Section 7.2.1, Paragraph 2; Section 7.2.2, + Paragraph 3; Section 7.2.3, Paragraph 5; Section 7.2.3, + Paragraph 5; Section 7.2.4, Paragraph 2; Section 7.2.4, + Paragraph 2; Section 7.2.4, Paragraph 3; Section 7.2.5, + Paragraph 8; Section 7.2.6, Paragraph 3; Section 7.2.6, + Paragraph 5; Section 7.2.7, Paragraph 2; Section 8.1, + Paragraph 2.22.1; Section 9, Paragraph 4; Appendix A.2.4, + Paragraph 3; Appendix A.3, Paragraph 1 + + D + + DATA Section 2, Paragraph 3; Section 4.1, Paragraph 5, Item 2; + Section 4.1, Paragraph 7; Section 4.1, Paragraph 7; + Section 4.1.2, Paragraph 3; Section 4.1.2, Paragraph 3; + Section 4.4, Paragraph 7; Section 4.4, Paragraph 7; + Section 4.4, Paragraph 7; Section 4.4, Paragraph 7; + Section 4.4, Paragraph 8; Section 4.6, Paragraph 12; + Table 1; *_Section 7.2.1_*; Table 2; Appendix A.1, Paragraph + 3; Appendix A.2.3, Paragraph 1; Appendix A.2.5 + + G + + GOAWAY Section 3.3, Paragraph 5; Section 5.2, Paragraph 1; + Section 5.2, Paragraph 1; Section 5.2, Paragraph 1; + Section 5.2, Paragraph 2; Section 5.2, Paragraph 2; + Section 5.2, Paragraph 3; Section 5.2, Paragraph 5.1.1; + Section 5.2, Paragraph 5.1.1; Section 5.2, Paragraph 5.1.2; + Section 5.2, Paragraph 5.1.2; Section 5.2, Paragraph 5, Item + 2; Section 5.2, Paragraph 5, Item 2; Section 5.2, Paragraph + 6; Section 5.2, Paragraph 6; Section 5.2, Paragraph 7; + Section 5.2, Paragraph 7; Section 5.2, Paragraph 8; + Section 5.2, Paragraph 8; Section 5.2, Paragraph 9; + Section 5.2, Paragraph 9; Section 5.2, Paragraph 10; + Section 5.2, Paragraph 12; Section 5.3, Paragraph 2; + Section 5.3, Paragraph 2; Section 5.4, Paragraph 2; Table 1; + *_Section 7.2.6_*; Table 2; Appendix A.2.5; Appendix A.2.5, + Paragraph 1.16.1 + + H + + H3_CLOSED_CRITICAL_STREAM Section 6.2.1, Paragraph 2; + Section 8.1; Table 4; Appendix A.4, Paragraph 3.4.1 + H3_CONNECT_ERROR Section 4.4, Paragraph 10; Section 8.1; + Table 4; Appendix A.4, Paragraph 3.22.1 + H3_EXCESSIVE_LOAD Section 8.1; Section 10.5, Paragraph 7; + Table 4; Appendix A.4, Paragraph 3.24.1 + H3_FRAME_ERROR Section 7.1, Paragraph 5; Section 7.1, + Paragraph 6; Section 8.1; Table 4; Appendix A.4, Paragraph + 3.14.1 + H3_FRAME_UNEXPECTED Section 4.1, Paragraph 7; Section 4.1, + Paragraph 8; Section 4.4, Paragraph 8; Section 7.2.1, + Paragraph 2; Section 7.2.2, Paragraph 3; Section 7.2.3, + Paragraph 5; Section 7.2.4, Paragraph 2; Section 7.2.4, + Paragraph 3; Section 7.2.5, Paragraph 8; Section 7.2.5, + Paragraph 9; Section 7.2.6, Paragraph 5; Section 7.2.7, + Paragraph 2; Section 7.2.7, Paragraph 3; Section 7.2.8, + Paragraph 3; Section 8.1; Table 4; Appendix A.4, Paragraph + 3.4.1 + H3_GENERAL_PROTOCOL_ERROR Section 7.2.5, Paragraph 6; + Section 8.1; Table 4; Appendix A.4, Paragraph 3.4.1 + H3_ID_ERROR Section 4.6, Paragraph 3; Section 5.2, Paragraph + 7; Section 6.2.2, Paragraph 6; Section 7.2.3, Paragraph 7; + Section 7.2.3, Paragraph 8; Section 7.2.5, Paragraph 5; + Section 7.2.6, Paragraph 3; Section 7.2.7, Paragraph 6; + Section 8.1; Table 4 + H3_INTERNAL_ERROR Section 8.1; Table 4; Appendix A.4, + Paragraph 3.6.1 + H3_MESSAGE_ERROR Section 4.1.2, Paragraph 4; Section 8.1; + Table 4; Appendix A.4, Paragraph 3.4.1 + H3_MISSING_SETTINGS Section 6.2.1, Paragraph 2; Section 8.1; + Table 4 + H3_NO_ERROR Section 4.1, Paragraph 15; Section 5.2, Paragraph + 11; Section 6.2.3, Paragraph 2; Section 8, Paragraph 5; + Section 8.1; Section 8.1, Paragraph 3; Section 8.1, + Paragraph 3; Table 4; Appendix A.4, Paragraph 3.2.1 + H3_REQUEST_CANCELLED Section 4.1.1, Paragraph 4; + Section 4.1.1, Paragraph 5; Section 4.6, Paragraph 14; + Section 7.2.3, Paragraph 3; Section 7.2.3, Paragraph 4; + Section 8.1; Table 4; Appendix A.4, Paragraph 3.18.1; + Appendix A.4.1, Paragraph 3 + H3_REQUEST_INCOMPLETE Section 4.1, Paragraph 14; Section 8.1; + Table 4 + H3_REQUEST_REJECTED Section 4.1.1, Paragraph 3; Section 4.1.1, + Paragraph 4; Section 4.1.1, Paragraph 5; Section 4.1.1, + Paragraph 5; Section 8.1; Table 4; Appendix A.4, Paragraph + 3.16.1; Appendix A.4.1, Paragraph 3 + H3_SETTINGS_ERROR Section 7.2.4, Paragraph 6; Section 7.2.4.1, + Paragraph 5; Section 7.2.4.2, Paragraph 8; Section 7.2.4.2, + Paragraph 8; Section 8.1; Table 4 + H3_STREAM_CREATION_ERROR Section 6.1, Paragraph 3; + Section 6.2, Paragraph 7; Section 6.2.1, Paragraph 2; + Section 6.2.2, Paragraph 3; Section 8.1; Table 4 + H3_VERSION_FALLBACK Section 8.1; Table 4; Appendix A.4, + Paragraph 3.28.1 + HEADERS Section 2, Paragraph 3; Section 4.1, Paragraph 5, Item + 1; Section 4.1, Paragraph 5, Item 3; Section 4.1, Paragraph + 7; Section 4.1, Paragraph 7; Section 4.1, Paragraph 7; + Section 4.1, Paragraph 10; Section 4.4, Paragraph 6; + Section 4.6, Paragraph 12; Table 1; *_Section 7.2.2_*; + Section 9, Paragraph 5; Table 2; Appendix A.2.1, Paragraph + 1; Appendix A.2.5; Appendix A.2.5, Paragraph 1.4.1; + Appendix A.2.5, Paragraph 1.20.1 + + M + + malformed Section 4.1, Paragraph 3; *_Section 4.1.2_*; + Section 4.2, Paragraph 2; Section 4.2, Paragraph 3; + Section 4.2, Paragraph 5; Section 4.3, Paragraph 3; + Section 4.3, Paragraph 4; Section 4.3.1, Paragraph 5; + Section 4.3.2, Paragraph 1; Section 4.4, Paragraph 5; + Section 8.1, Paragraph 2.30.1; Section 10.3, Paragraph 1; + Section 10.3, Paragraph 2; Section 10.5.1, Paragraph 2 + MAX_PUSH_ID Section 2, Paragraph 5; Section 4.6, Paragraph 3; + Section 4.6, Paragraph 3; Section 4.6, Paragraph 3; + Section 4.6, Paragraph 3; Table 1; Section 7.2.5, Paragraph + 5; *_Section 7.2.7_*; Table 2; Appendix A.1, Paragraph 4; + Appendix A.3, Paragraph 4.4.1 + + P + + push ID *_Section 4.6_*; Section 5.2, Paragraph 1; + Section 5.2, Paragraph 5, Item 2; Section 5.2, Paragraph 9; + Section 6.2.2, Paragraph 2; Section 6.2.2, Paragraph 6; + Section 6.2.2, Paragraph 6; Section 7.2.3, Paragraph 1; + Section 7.2.3, Paragraph 7; Section 7.2.3, Paragraph 7; + Section 7.2.3, Paragraph 8; Section 7.2.3, Paragraph 8; + Section 7.2.5, Paragraph 4.2.1; Section 7.2.5, Paragraph 5; + Section 7.2.5, Paragraph 5; Section 7.2.5, Paragraph 6; + Section 7.2.5, Paragraph 6; Section 7.2.5, Paragraph 7; + Section 7.2.5, Paragraph 7; Section 7.2.5, Paragraph 7; + Section 7.2.6, Paragraph 4; Section 7.2.7, Paragraph 1; + Section 7.2.7, Paragraph 4; Section 7.2.7, Paragraph 4; + Section 7.2.7, Paragraph 6; Section 7.2.7, Paragraph 6; + Section 8.1, Paragraph 2.18.1; Appendix A.2.5, Paragraph + 1.12.1; Appendix A.2.5, Paragraph 1.16.1 + push stream Section 4.1, Paragraph 8; Section 4.1, Paragraph + 9; Section 4.6, Paragraph 3; Section 4.6, Paragraph 5; + Section 4.6, Paragraph 5; Section 4.6, Paragraph 13; + Section 4.6, Paragraph 13; Section 4.6, Paragraph 13; + Section 6.2, Paragraph 3; *_Section 6.2.2_*; Section 7, + Paragraph 1; Section 7.2.2, Paragraph 3; Section 7.2.3, + Paragraph 1; Section 7.2.3, Paragraph 2; Section 7.2.3, + Paragraph 2; Section 7.2.3, Paragraph 2; Section 7.2.3, + Paragraph 2; Section 7.2.3, Paragraph 3; Section 7.2.3, + Paragraph 4; Section 7.2.3, Paragraph 4; Section 7.2.3, + Paragraph 4; Section 7.2.5, Paragraph 4.2.1; Section 7.2.7, + Paragraph 1; Appendix A.2.5, Paragraph 1.12.1 + PUSH_PROMISE Section 2, Paragraph 5; Section 4.1, Paragraph 8; + Section 4.1, Paragraph 8; Section 4.1, Paragraph 8; + Section 4.1, Paragraph 8; Section 4.1, Paragraph 10; + Section 4.6, Paragraph 4; Section 4.6, Paragraph 10; + Section 4.6, Paragraph 11; Section 4.6, Paragraph 11; + Section 4.6, Paragraph 12; Section 4.6, Paragraph 12; + Section 4.6, Paragraph 13; Section 4.6, Paragraph 13; + Section 4.6, Paragraph 13; Table 1; Section 7.2.3, Paragraph + 8; Section 7.2.3, Paragraph 8; *_Section 7.2.5_*; + Section 7.2.7, Paragraph 1; Section 10.4, Paragraph 1; + Section 10.5, Paragraph 2; Table 2; Appendix A.2.5; + Appendix A.2.5, Paragraph 1.12.1; Appendix A.2.5, Paragraph + 1.12.1; Appendix A.2.5, Paragraph 1.20.1 + + R + + request stream Section 4.1, Paragraph 1; Section 4.1, + Paragraph 15; Section 4.1, Paragraph 15; Section 4.1.1, + Paragraph 1; Section 4.1.1, Paragraph 5; Section 4.4, + Paragraph 5; Section 4.4, Paragraph 9; Section 4.6, + Paragraph 4; Section 4.6, Paragraph 4; Section 4.6, + Paragraph 11; Section 4.6, Paragraph 11; *_Section 6.1_*; + Section 7, Paragraph 1; Section 7.2.2, Paragraph 3; + Section 7.2.5, Paragraph 1 + + S + + SETTINGS Section 3.2, Paragraph 4; Section 3.2, Paragraph 4; + Section 6.2.1, Paragraph 2; Table 1; Section 7, Paragraph 3; + *_Section 7.2.4_*; Section 8.1, Paragraph 2.20.1; + Section 8.1, Paragraph 2.22.1; Section 9, Paragraph 4; + Section 10.5, Paragraph 4; Table 2; Table 4; Table 4; + Appendix A.2.5; Appendix A.2.5, Paragraph 1.10.1; + Appendix A.3, Paragraph 2; Appendix A.3, Paragraph 3; + Appendix A.3, Paragraph 4.4.1; Appendix A.3, Paragraph + 4.6.1; Appendix A.3, Paragraph 4.8.1; Appendix A.3, + Paragraph 4.10.1; Appendix A.4, Paragraph 3.10.1 + SETTINGS_MAX_FIELD_SECTION_SIZE Section 4.2.2, Paragraph 2; + Section 7.2.4.1; Section 10.5.1, Paragraph 2; Appendix A.3, + Paragraph 4.12.1 + stream error Section 2.2; Section 4.1.2, Paragraph 4; + Section 4.4, Paragraph 10; *_Section 8_*; Appendix A.4.1, + Paragraph 3; Appendix A.4.1, Paragraph 3; Appendix A.4.1, + Paragraph 4 + +Author's Address + + Mike Bishop (editor) + Akamai + Email: mbishop@evequefou.be diff --git a/eval/corpora/rfc/RFC 9204 - QPACK Field Compression for HTTP3.txt b/eval/corpora/rfc/RFC 9204 - QPACK Field Compression for HTTP3.txt new file mode 100644 index 00000000..d9aa0f68 --- /dev/null +++ b/eval/corpora/rfc/RFC 9204 - QPACK Field Compression for HTTP3.txt @@ -0,0 +1,2133 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) C. Krasic +Request for Comments: 9204 +Category: Standards Track M. Bishop +ISSN: 2070-1721 Akamai Technologies + A. Frindell, Ed. + Facebook + June 2022 + + + QPACK: Field Compression for HTTP/3 + +Abstract + + This specification defines QPACK: a compression format for + efficiently representing HTTP fields that is to be used in HTTP/3. + This is a variation of HPACK compression that seeks to reduce head- + of-line blocking. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9204. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 1.1. Conventions and Definitions + 1.2. Notational Conventions + 2. Compression Process Overview + 2.1. Encoder + 2.1.1. Limits on Dynamic Table Insertions + 2.1.2. Blocked Streams + 2.1.3. Avoiding Flow-Control Deadlocks + 2.1.4. Known Received Count + 2.2. Decoder + 2.2.1. Blocked Decoding + 2.2.2. State Synchronization + 2.2.3. Invalid References + 3. Reference Tables + 3.1. Static Table + 3.2. Dynamic Table + 3.2.1. Dynamic Table Size + 3.2.2. Dynamic Table Capacity and Eviction + 3.2.3. Maximum Dynamic Table Capacity + 3.2.4. Absolute Indexing + 3.2.5. Relative Indexing + 3.2.6. Post-Base Indexing + 4. Wire Format + 4.1. Primitives + 4.1.1. Prefixed Integers + 4.1.2. String Literals + 4.2. Encoder and Decoder Streams + 4.3. Encoder Instructions + 4.3.1. Set Dynamic Table Capacity + 4.3.2. Insert with Name Reference + 4.3.3. Insert with Literal Name + 4.3.4. Duplicate + 4.4. Decoder Instructions + 4.4.1. Section Acknowledgment + 4.4.2. Stream Cancellation + 4.4.3. Insert Count Increment + 4.5. Field Line Representations + 4.5.1. Encoded Field Section Prefix + 4.5.2. Indexed Field Line + 4.5.3. Indexed Field Line with Post-Base Index + 4.5.4. Literal Field Line with Name Reference + 4.5.5. Literal Field Line with Post-Base Name Reference + 4.5.6. Literal Field Line with Literal Name + 5. Configuration + 6. Error Handling + 7. Security Considerations + 7.1. Probing Dynamic Table State + 7.1.1. Applicability to QPACK and HTTP + 7.1.2. Mitigation + 7.1.3. Never-Indexed Literals + 7.2. Static Huffman Encoding + 7.3. Memory Consumption + 7.4. Implementation Limits + 8. IANA Considerations + 8.1. Settings Registration + 8.2. Stream Type Registration + 8.3. Error Code Registration + 9. References + 9.1. Normative References + 9.2. Informative References + Appendix A. Static Table + Appendix B. Encoding and Decoding Examples + B.1. Literal Field Line with Name Reference + B.2. Dynamic Table + B.3. Speculative Insert + B.4. Duplicate Instruction, Stream Cancellation + B.5. Dynamic Table Insert, Eviction + Appendix C. Sample Single-Pass Encoding Algorithm + Acknowledgments + Authors' Addresses + +1. Introduction + + The QUIC transport protocol ([QUIC-TRANSPORT]) is designed to support + HTTP semantics, and its design subsumes many of the features of + HTTP/2 ([HTTP/2]). HTTP/2 uses HPACK ([RFC7541]) for compression of + the header and trailer sections. If HPACK were used for HTTP/3 + ([HTTP/3]), it would induce head-of-line blocking for field sections + due to built-in assumptions of a total ordering across frames on all + streams. + + QPACK reuses core concepts from HPACK, but is redesigned to allow + correctness in the presence of out-of-order delivery, with + flexibility for implementations to balance between resilience against + head-of-line blocking and optimal compression ratio. The design + goals are to closely approach the compression ratio of HPACK with + substantially less head-of-line blocking under the same loss + conditions. + +1.1. Conventions and Definitions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + The following terms are used in this document: + + HTTP fields: Metadata sent as part of an HTTP message. The term + encompasses both header and trailer fields. Colloquially, the + term "headers" has often been used to refer to HTTP header fields + and trailer fields; this document uses "fields" for generality. + + HTTP field line: A name-value pair sent as part of an HTTP field + section. See Sections 6.3 and 6.5 of [HTTP]. + + HTTP field value: Data associated with a field name, composed from + all field line values with that field name in that section, + concatenated together with comma separators. + + Field section: An ordered collection of HTTP field lines associated + with an HTTP message. A field section can contain multiple field + lines with the same name. It can also contain duplicate field + lines. An HTTP message can include both header and trailer + sections. + + Representation: An instruction that represents a field line, + possibly by reference to the dynamic and static tables. + + Encoder: An implementation that encodes field sections. + + Decoder: An implementation that decodes encoded field sections. + + Absolute Index: A unique index for each entry in the dynamic table. + + Base: A reference point for relative and post-Base indices. + Representations that reference dynamic table entries are relative + to a Base. + + Insert Count: The total number of entries inserted in the dynamic + table. + + Note that QPACK is a name, not an abbreviation. + +1.2. Notational Conventions + + Diagrams in this document use the format described in Section 3.1 of + [RFC2360], with the following additional conventions: + + x (A) Indicates that x is A bits long. + + x (A+) Indicates that x uses the prefixed integer encoding defined + in Section 4.1.1, beginning with an A-bit prefix. + + x ... Indicates that x is variable length and extends to the end of + the region. + +2. Compression Process Overview + + Like HPACK, QPACK uses two tables for associating field lines + ("headers") to indices. The static table (Section 3.1) is predefined + and contains common header field lines (some of them with an empty + value). The dynamic table (Section 3.2) is built up over the course + of the connection and can be used by the encoder to index both header + and trailer field lines in the encoded field sections. + + QPACK defines unidirectional streams for sending instructions from + encoder to decoder and vice versa. + +2.1. Encoder + + An encoder converts a header or trailer section into a series of + representations by emitting either an indexed or a literal + representation for each field line in the list; see Section 4.5. + Indexed representations achieve high compression by replacing the + literal name and possibly the value with an index to either the + static or dynamic table. References to the static table and literal + representations do not require any dynamic state and never risk head- + of-line blocking. References to the dynamic table risk head-of-line + blocking if the encoder has not received an acknowledgment indicating + the entry is available at the decoder. + + An encoder MAY insert any entry in the dynamic table it chooses; it + is not limited to field lines it is compressing. + + QPACK preserves the ordering of field lines within each field + section. An encoder MUST emit field representations in the order + they appear in the input field section. + + QPACK is designed to place the burden of optional state tracking on + the encoder, resulting in relatively simple decoders. + +2.1.1. Limits on Dynamic Table Insertions + + Inserting entries into the dynamic table might not be possible if the + table contains entries that cannot be evicted. + + A dynamic table entry cannot be evicted immediately after insertion, + even if it has never been referenced. Once the insertion of a + dynamic table entry has been acknowledged and there are no + outstanding references to the entry in unacknowledged + representations, the entry becomes evictable. Note that references + on the encoder stream never preclude the eviction of an entry, + because those references are guaranteed to be processed before the + instruction evicting the entry. + + If the dynamic table does not contain enough room for a new entry + without evicting other entries, and the entries that would be evicted + are not evictable, the encoder MUST NOT insert that entry into the + dynamic table (including duplicates of existing entries). In order + to avoid this, an encoder that uses the dynamic table has to keep + track of each dynamic table entry referenced by each field section + until those representations are acknowledged by the decoder; see + Section 4.4.1. + +2.1.1.1. Avoiding Prohibited Insertions + + To ensure that the encoder is not prevented from adding new entries, + the encoder can avoid referencing entries that are close to eviction. + Rather than reference such an entry, the encoder can emit a Duplicate + instruction (Section 4.3.4) and reference the duplicate instead. + + Determining which entries are too close to eviction to reference is + an encoder preference. One heuristic is to target a fixed amount of + available space in the dynamic table: either unused space or space + that can be reclaimed by evicting non-blocking entries. To achieve + this, the encoder can maintain a draining index, which is the + smallest absolute index (Section 3.2.4) in the dynamic table that it + will emit a reference for. As new entries are inserted, the encoder + increases the draining index to maintain the section of the table + that it will not reference. If the encoder does not create new + references to entries with an absolute index lower than the draining + index, the number of unacknowledged references to those entries will + eventually become zero, allowing them to be evicted. + + <-- Newer Entries Older Entries --> + (Larger Indices) (Smaller Indices) + +--------+---------------------------------+----------+ + | Unused | Referenceable | Draining | + | Space | Entries | Entries | + +--------+---------------------------------+----------+ + ^ ^ ^ + | | | + Insertion Point Draining Index Dropping + Point + + Figure 1: Draining Dynamic Table Entries + +2.1.2. Blocked Streams + + Because QUIC does not guarantee order between data on different + streams, a decoder might encounter a representation that references a + dynamic table entry that it has not yet received. + + Each encoded field section contains a Required Insert Count + (Section 4.5.1), the lowest possible value for the Insert Count with + which the field section can be decoded. For a field section encoded + using references to the dynamic table, the Required Insert Count is + one larger than the largest absolute index of all referenced dynamic + table entries. For a field section encoded with no references to the + dynamic table, the Required Insert Count is zero. + + When the decoder receives an encoded field section with a Required + Insert Count greater than its own Insert Count, the stream cannot be + processed immediately and is considered "blocked"; see Section 2.2.1. + + The decoder specifies an upper bound on the number of streams that + can be blocked using the SETTINGS_QPACK_BLOCKED_STREAMS setting; see + Section 5. An encoder MUST limit the number of streams that could + become blocked to the value of SETTINGS_QPACK_BLOCKED_STREAMS at all + times. If a decoder encounters more blocked streams than it promised + to support, it MUST treat this as a connection error of type + QPACK_DECOMPRESSION_FAILED. + + Note that the decoder might not become blocked on every stream that + risks becoming blocked. + + An encoder can decide whether to risk having a stream become blocked. + If permitted by the value of SETTINGS_QPACK_BLOCKED_STREAMS, + compression efficiency can often be improved by referencing dynamic + table entries that are still in transit, but if there is loss or + reordering, the stream can become blocked at the decoder. An encoder + can avoid the risk of blocking by only referencing dynamic table + entries that have been acknowledged, but this could mean using + literals. Since literals make the encoded field section larger, this + can result in the encoder becoming blocked on congestion or flow- + control limits. + +2.1.3. Avoiding Flow-Control Deadlocks + + Writing instructions on streams that are limited by flow control can + produce deadlocks. + + A decoder might stop issuing flow-control credit on the stream that + carries an encoded field section until the necessary updates are + received on the encoder stream. If the granting of flow-control + credit on the encoder stream (or the connection as a whole) depends + on the consumption and release of data on the stream carrying the + encoded field section, a deadlock might result. + + More generally, a stream containing a large instruction can become + deadlocked if the decoder withholds flow-control credit until the + instruction is completely received. + + To avoid these deadlocks, an encoder SHOULD NOT write an instruction + unless sufficient stream and connection flow-control credit is + available for the entire instruction. + +2.1.4. Known Received Count + + The Known Received Count is the total number of dynamic table + insertions and duplications acknowledged by the decoder. The encoder + tracks the Known Received Count in order to identify which dynamic + table entries can be referenced without potentially blocking a + stream. The decoder tracks the Known Received Count in order to be + able to send Insert Count Increment instructions. + + A Section Acknowledgment instruction (Section 4.4.1) implies that the + decoder has received all dynamic table state necessary to decode the + field section. If the Required Insert Count of the acknowledged + field section is greater than the current Known Received Count, the + Known Received Count is updated to that Required Insert Count value. + + An Insert Count Increment instruction (Section 4.4.3) increases the + Known Received Count by its Increment parameter. See Section 2.2.2.3 + for guidance. + +2.2. Decoder + + As in HPACK, the decoder processes a series of representations and + emits the corresponding field sections. It also processes + instructions received on the encoder stream that modify the dynamic + table. Note that encoded field sections and encoder stream + instructions arrive on separate streams. This is unlike HPACK, where + encoded field sections (header blocks) can contain instructions that + modify the dynamic table, and there is no dedicated stream of HPACK + instructions. + + The decoder MUST emit field lines in the order their representations + appear in the encoded field section. + +2.2.1. Blocked Decoding + + Upon receipt of an encoded field section, the decoder examines the + Required Insert Count. When the Required Insert Count is less than + or equal to the decoder's Insert Count, the field section can be + processed immediately. Otherwise, the stream on which the field + section was received becomes blocked. + + While blocked, encoded field section data SHOULD remain in the + blocked stream's flow-control window. This data is unusable until + the stream becomes unblocked, and releasing the flow control + prematurely makes the decoder vulnerable to memory exhaustion + attacks. A stream becomes unblocked when the Insert Count becomes + greater than or equal to the Required Insert Count for all encoded + field sections the decoder has started reading from the stream. + + When processing encoded field sections, the decoder expects the + Required Insert Count to equal the lowest possible value for the + Insert Count with which the field section can be decoded, as + prescribed in Section 2.1.2. If it encounters a Required Insert + Count smaller than expected, it MUST treat this as a connection error + of type QPACK_DECOMPRESSION_FAILED; see Section 2.2.3. If it + encounters a Required Insert Count larger than expected, it MAY treat + this as a connection error of type QPACK_DECOMPRESSION_FAILED. + +2.2.2. State Synchronization + + The decoder signals the following events by emitting decoder + instructions (Section 4.4) on the decoder stream. + +2.2.2.1. Completed Processing of a Field Section + + After the decoder finishes decoding a field section encoded using + representations containing dynamic table references, it MUST emit a + Section Acknowledgment instruction (Section 4.4.1). A stream may + carry multiple field sections in the case of intermediate responses, + trailers, and pushed requests. The encoder interprets each + Section Acknowledgment instruction as acknowledging the earliest + unacknowledged field section containing dynamic table references sent + on the given stream. + +2.2.2.2. Abandonment of a Stream + + When an endpoint receives a stream reset before the end of a stream + or before all encoded field sections are processed on that stream, or + when it abandons reading of a stream, it generates a Stream + Cancellation instruction; see Section 4.4.2. This signals to the + encoder that all references to the dynamic table on that stream are + no longer outstanding. A decoder with a maximum dynamic table + capacity (Section 3.2.3) equal to zero MAY omit sending Stream + Cancellations, because the encoder cannot have any dynamic table + references. An encoder cannot infer from this instruction that any + updates to the dynamic table have been received. + + The Section Acknowledgment and Stream Cancellation instructions + permit the encoder to remove references to entries in the dynamic + table. When an entry with an absolute index lower than the Known + Received Count has zero references, then it is considered evictable; + see Section 2.1.1. + +2.2.2.3. New Table Entries + + After receiving new table entries on the encoder stream, the decoder + chooses when to emit Insert Count Increment instructions; see + Section 4.4.3. Emitting this instruction after adding each new + dynamic table entry will provide the timeliest feedback to the + encoder, but could be redundant with other decoder feedback. By + delaying an Insert Count Increment instruction, the decoder might be + able to coalesce multiple Insert Count Increment instructions or + replace them entirely with Section Acknowledgments; see + Section 4.4.1. However, delaying too long may lead to compression + inefficiencies if the encoder waits for an entry to be acknowledged + before using it. + +2.2.3. Invalid References + + If the decoder encounters a reference in a field line representation + to a dynamic table entry that has already been evicted or that has an + absolute index greater than or equal to the declared Required Insert + Count (Section 4.5.1), it MUST treat this as a connection error of + type QPACK_DECOMPRESSION_FAILED. + + If the decoder encounters a reference in an encoder instruction to a + dynamic table entry that has already been evicted, it MUST treat this + as a connection error of type QPACK_ENCODER_STREAM_ERROR. + +3. Reference Tables + + Unlike in HPACK, entries in the QPACK static and dynamic tables are + addressed separately. The following sections describe how entries in + each table are addressed. + +3.1. Static Table + + The static table consists of a predefined list of field lines, each + of which has a fixed index over time. Its entries are defined in + Appendix A. + + All entries in the static table have a name and a value. However, + values can be empty (that is, have a length of 0). Each entry is + identified by a unique index. + + Note that the QPACK static table is indexed from 0, whereas the HPACK + static table is indexed from 1. + + When the decoder encounters an invalid static table index in a field + line representation, it MUST treat this as a connection error of type + QPACK_DECOMPRESSION_FAILED. If this index is received on the encoder + stream, this MUST be treated as a connection error of type + QPACK_ENCODER_STREAM_ERROR. + +3.2. Dynamic Table + + The dynamic table consists of a list of field lines maintained in + first-in, first-out order. A QPACK encoder and decoder share a + dynamic table that is initially empty. The encoder adds entries to + the dynamic table and sends them to the decoder via instructions on + the encoder stream; see Section 4.3. + + The dynamic table can contain duplicate entries (i.e., entries with + the same name and same value). Therefore, duplicate entries MUST NOT + be treated as an error by the decoder. + + Dynamic table entries can have empty values. + +3.2.1. Dynamic Table Size + + The size of the dynamic table is the sum of the size of its entries. + + The size of an entry is the sum of its name's length in bytes, its + value's length in bytes, and 32 additional bytes. The size of an + entry is calculated using the length of its name and value without + Huffman encoding applied. + +3.2.2. Dynamic Table Capacity and Eviction + + The encoder sets the capacity of the dynamic table, which serves as + the upper limit on its size. The initial capacity of the dynamic + table is zero. The encoder sends a Set Dynamic Table Capacity + instruction (Section 4.3.1) with a non-zero capacity to begin using + the dynamic table. + + Before a new entry is added to the dynamic table, entries are evicted + from the end of the dynamic table until the size of the dynamic table + is less than or equal to (table capacity - size of new entry). The + encoder MUST NOT cause a dynamic table entry to be evicted unless + that entry is evictable; see Section 2.1.1. The new entry is then + added to the table. It is an error if the encoder attempts to add an + entry that is larger than the dynamic table capacity; the decoder + MUST treat this as a connection error of type + QPACK_ENCODER_STREAM_ERROR. + + A new entry can reference an entry in the dynamic table that will be + evicted when adding this new entry into the dynamic table. + Implementations are cautioned to avoid deleting the referenced name + or value if the referenced entry is evicted from the dynamic table + prior to inserting the new entry. + + Whenever the dynamic table capacity is reduced by the encoder + (Section 4.3.1), entries are evicted from the end of the dynamic + table until the size of the dynamic table is less than or equal to + the new table capacity. This mechanism can be used to completely + clear entries from the dynamic table by setting a capacity of 0, + which can subsequently be restored. + +3.2.3. Maximum Dynamic Table Capacity + + To bound the memory requirements of the decoder, the decoder limits + the maximum value the encoder is permitted to set for the dynamic + table capacity. In HTTP/3, this limit is determined by the value of + SETTINGS_QPACK_MAX_TABLE_CAPACITY sent by the decoder; see Section 5. + The encoder MUST NOT set a dynamic table capacity that exceeds this + maximum, but it can choose to use a lower dynamic table capacity; see + Section 4.3.1. + + For clients using 0-RTT data in HTTP/3, the server's maximum table + capacity is the remembered value of the setting or zero if the value + was not previously sent. When the client's 0-RTT value of the + SETTING is zero, the server MAY set it to a non-zero value in its + SETTINGS frame. If the remembered value is non-zero, the server MUST + send the same non-zero value in its SETTINGS frame. If it specifies + any other value, or omits SETTINGS_QPACK_MAX_TABLE_CAPACITY from + SETTINGS, the encoder must treat this as a connection error of type + QPACK_DECODER_STREAM_ERROR. + + For clients not using 0-RTT data (whether 0-RTT is not attempted or + is rejected) and for all HTTP/3 servers, the maximum table capacity + is 0 until the encoder processes a SETTINGS frame with a non-zero + value of SETTINGS_QPACK_MAX_TABLE_CAPACITY. + + When the maximum table capacity is zero, the encoder MUST NOT insert + entries into the dynamic table and MUST NOT send any encoder + instructions on the encoder stream. + +3.2.4. Absolute Indexing + + Each entry possesses an absolute index that is fixed for the lifetime + of that entry. The first entry inserted has an absolute index of 0; + indices increase by one with each insertion. + +3.2.5. Relative Indexing + + Relative indices begin at zero and increase in the opposite direction + from the absolute index. Determining which entry has a relative + index of 0 depends on the context of the reference. + + In encoder instructions (Section 4.3), a relative index of 0 refers + to the most recently inserted value in the dynamic table. Note that + this means the entry referenced by a given relative index will change + while interpreting instructions on the encoder stream. + + +-----+---------------+-------+ + | n-1 | ... | d | Absolute Index + + - - +---------------+ - - - + + | 0 | ... | n-d-1 | Relative Index + +-----+---------------+-------+ + ^ | + | V + Insertion Point Dropping Point + + n = count of entries inserted + d = count of entries dropped + + Figure 2: Example Dynamic Table Indexing - Encoder Stream + + Unlike in encoder instructions, relative indices in field line + representations are relative to the Base at the beginning of the + encoded field section; see Section 4.5.1. This ensures that + references are stable even if encoded field sections and dynamic + table updates are processed out of order. + + In a field line representation, a relative index of 0 refers to the + entry with absolute index equal to Base - 1. + + Base + | + V + +-----+-----+-----+-----+-------+ + | n-1 | n-2 | n-3 | ... | d | Absolute Index + +-----+-----+ - +-----+ - + + | 0 | ... | n-d-3 | Relative Index + +-----+-----+-------+ + + n = count of entries inserted + d = count of entries dropped + In this example, Base = n - 2 + + Figure 3: Example Dynamic Table Indexing - Relative Index in + Representation + +3.2.6. Post-Base Indexing + + Post-Base indices are used in field line representations for entries + with absolute indices greater than or equal to Base, starting at 0 + for the entry with absolute index equal to Base and increasing in the + same direction as the absolute index. + + Post-Base indices allow an encoder to process a field section in a + single pass and include references to entries added while processing + this (or other) field sections. + + Base + | + V + +-----+-----+-----+-----+-----+ + | n-1 | n-2 | n-3 | ... | d | Absolute Index + +-----+-----+-----+-----+-----+ + | 1 | 0 | Post-Base Index + +-----+-----+ + + n = count of entries inserted + d = count of entries dropped + In this example, Base = n - 2 + + Figure 4: Example Dynamic Table Indexing - Post-Base Index in + Representation + +4. Wire Format + +4.1. Primitives + +4.1.1. Prefixed Integers + + The prefixed integer from Section 5.1 of [RFC7541] is used heavily + throughout this document. The format from [RFC7541] is used + unmodified. Note, however, that QPACK uses some prefix sizes not + actually used in HPACK. + + QPACK implementations MUST be able to decode integers up to and + including 62 bits long. + +4.1.2. String Literals + + The string literal defined by Section 5.2 of [RFC7541] is also used + throughout. This string format includes optional Huffman encoding. + + HPACK defines string literals to begin on a byte boundary. They + begin with a single bit flag, denoted as 'H' in this document + (indicating whether the string is Huffman encoded), followed by the + string length encoded as a 7-bit prefix integer, and finally the + indicated number of bytes of data. When Huffman encoding is enabled, + the Huffman table from Appendix B of [RFC7541] is used without + modification and the indicated length is the size of the string after + encoding. + + This document expands the definition of string literals by permitting + them to begin other than on a byte boundary. An "N-bit prefix string + literal" begins mid-byte, with the first (8-N) bits allocated to a + previous field. The string uses one bit for the Huffman flag, + followed by the length of the encoded string as a (N-1)-bit prefix + integer. The prefix size, N, can have a value between 2 and 8, + inclusive. The remainder of the string literal is unmodified. + + A string literal without a prefix length noted is an 8-bit prefix + string literal and follows the definitions in [RFC7541] without + modification. + +4.2. Encoder and Decoder Streams + + QPACK defines two unidirectional stream types: + + * An encoder stream is a unidirectional stream of type 0x02. It + carries an unframed sequence of encoder instructions from encoder + to decoder. + + * A decoder stream is a unidirectional stream of type 0x03. It + carries an unframed sequence of decoder instructions from decoder + to encoder. + + HTTP/3 endpoints contain a QPACK encoder and decoder. Each endpoint + MUST initiate, at most, one encoder stream and, at most, one decoder + stream. Receipt of a second instance of either stream type MUST be + treated as a connection error of type H3_STREAM_CREATION_ERROR. + + The sender MUST NOT close either of these streams, and the receiver + MUST NOT request that the sender close either of these streams. + Closure of either unidirectional stream type MUST be treated as a + connection error of type H3_CLOSED_CRITICAL_STREAM. + + An endpoint MAY avoid creating an encoder stream if it will not be + used (for example, if its encoder does not wish to use the dynamic + table or if the maximum size of the dynamic table permitted by the + peer is zero). + + An endpoint MAY avoid creating a decoder stream if its decoder sets + the maximum capacity of the dynamic table to zero. + + An endpoint MUST allow its peer to create an encoder stream and a + decoder stream even if the connection's settings prevent their use. + +4.3. Encoder Instructions + + An encoder sends encoder instructions on the encoder stream to set + the capacity of the dynamic table and add dynamic table entries. + Instructions adding table entries can use existing entries to avoid + transmitting redundant information. The name can be transmitted as a + reference to an existing entry in the static or the dynamic table or + as a string literal. For entries that already exist in the dynamic + table, the full entry can also be used by reference, creating a + duplicate entry. + +4.3.1. Set Dynamic Table Capacity + + An encoder informs the decoder of a change to the dynamic table + capacity using an instruction that starts with the '001' 3-bit + pattern. This is followed by the new dynamic table capacity + represented as an integer with a 5-bit prefix; see Section 4.1.1. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 0 | 0 | 1 | Capacity (5+) | + +---+---+---+-------------------+ + + Figure 5: Set Dynamic Table Capacity + + The new capacity MUST be lower than or equal to the limit described + in Section 3.2.3. In HTTP/3, this limit is the value of the + SETTINGS_QPACK_MAX_TABLE_CAPACITY parameter (Section 5) received from + the decoder. The decoder MUST treat a new dynamic table capacity + value that exceeds this limit as a connection error of type + QPACK_ENCODER_STREAM_ERROR. + + Reducing the dynamic table capacity can cause entries to be evicted; + see Section 3.2.2. This MUST NOT cause the eviction of entries that + are not evictable; see Section 2.1.1. Changing the capacity of the + dynamic table is not acknowledged as this instruction does not insert + an entry. + +4.3.2. Insert with Name Reference + + An encoder adds an entry to the dynamic table where the field name + matches the field name of an entry stored in the static or the + dynamic table using an instruction that starts with the '1' 1-bit + pattern. The second ('T') bit indicates whether the reference is to + the static or dynamic table. The 6-bit prefix integer + (Section 4.1.1) that follows is used to locate the table entry for + the field name. When T=1, the number represents the static table + index; when T=0, the number is the relative index of the entry in the + dynamic table. + + The field name reference is followed by the field value represented + as a string literal; see Section 4.1.2. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 1 | T | Name Index (6+) | + +---+---+-----------------------+ + | H | Value Length (7+) | + +---+---------------------------+ + | Value String (Length bytes) | + +-------------------------------+ + + Figure 6: Insert Field Line -- Indexed Name + +4.3.3. Insert with Literal Name + + An encoder adds an entry to the dynamic table where both the field + name and the field value are represented as string literals using an + instruction that starts with the '01' 2-bit pattern. + + This is followed by the name represented as a 6-bit prefix string + literal and the value represented as an 8-bit prefix string literal; + see Section 4.1.2. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 0 | 1 | H | Name Length (5+) | + +---+---+---+-------------------+ + | Name String (Length bytes) | + +---+---------------------------+ + | H | Value Length (7+) | + +---+---------------------------+ + | Value String (Length bytes) | + +-------------------------------+ + + Figure 7: Insert Field Line -- New Name + +4.3.4. Duplicate + + An encoder duplicates an existing entry in the dynamic table using an + instruction that starts with the '000' 3-bit pattern. This is + followed by the relative index of the existing entry represented as + an integer with a 5-bit prefix; see Section 4.1.1. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 0 | 0 | 0 | Index (5+) | + +---+---+---+-------------------+ + + Figure 8: Duplicate + + The existing entry is reinserted into the dynamic table without + resending either the name or the value. This is useful to avoid + adding a reference to an older entry, which might block inserting new + entries. + +4.4. Decoder Instructions + + A decoder sends decoder instructions on the decoder stream to inform + the encoder about the processing of field sections and table updates + to ensure consistency of the dynamic table. + +4.4.1. Section Acknowledgment + + After processing an encoded field section whose declared Required + Insert Count is not zero, the decoder emits a Section Acknowledgment + instruction. The instruction starts with the '1' 1-bit pattern, + followed by the field section's associated stream ID encoded as a + 7-bit prefix integer; see Section 4.1.1. + + This instruction is used as described in Sections 2.1.4 and 2.2.2. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 1 | Stream ID (7+) | + +---+---------------------------+ + + Figure 9: Section Acknowledgment + + If an encoder receives a Section Acknowledgment instruction referring + to a stream on which every encoded field section with a non-zero + Required Insert Count has already been acknowledged, this MUST be + treated as a connection error of type QPACK_DECODER_STREAM_ERROR. + + The Section Acknowledgment instruction might increase the Known + Received Count; see Section 2.1.4. + +4.4.2. Stream Cancellation + + When a stream is reset or reading is abandoned, the decoder emits a + Stream Cancellation instruction. The instruction starts with the + '01' 2-bit pattern, followed by the stream ID of the affected stream + encoded as a 6-bit prefix integer. + + This instruction is used as described in Section 2.2.2. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 0 | 1 | Stream ID (6+) | + +---+---+-----------------------+ + + Figure 10: Stream Cancellation + +4.4.3. Insert Count Increment + + The Insert Count Increment instruction starts with the '00' 2-bit + pattern, followed by the Increment encoded as a 6-bit prefix integer. + This instruction increases the Known Received Count (Section 2.1.4) + by the value of the Increment parameter. The decoder should send an + Increment value that increases the Known Received Count to the total + number of dynamic table insertions and duplications processed so far. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 0 | 0 | Increment (6+) | + +---+---+-----------------------+ + + Figure 11: Insert Count Increment + + An encoder that receives an Increment field equal to zero, or one + that increases the Known Received Count beyond what the encoder has + sent, MUST treat this as a connection error of type + QPACK_DECODER_STREAM_ERROR. + +4.5. Field Line Representations + + An encoded field section consists of a prefix and a possibly empty + sequence of representations defined in this section. Each + representation corresponds to a single field line. These + representations reference the static table or the dynamic table in a + particular state, but they do not modify that state. + + Encoded field sections are carried in frames on streams defined by + the enclosing protocol. + +4.5.1. Encoded Field Section Prefix + + Each encoded field section is prefixed with two integers. The + Required Insert Count is encoded as an integer with an 8-bit prefix + using the encoding described in Section 4.5.1.1. The Base is encoded + as a Sign bit ('S') and a Delta Base value with a 7-bit prefix; see + Section 4.5.1.2. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | Required Insert Count (8+) | + +---+---------------------------+ + | S | Delta Base (7+) | + +---+---------------------------+ + | Encoded Field Lines ... + +-------------------------------+ + + Figure 12: Encoded Field Section + +4.5.1.1. Required Insert Count + + Required Insert Count identifies the state of the dynamic table + needed to process the encoded field section. Blocking decoders use + the Required Insert Count to determine when it is safe to process the + rest of the field section. + + The encoder transforms the Required Insert Count as follows before + encoding: + + if ReqInsertCount == 0: + EncInsertCount = 0 + else: + EncInsertCount = (ReqInsertCount mod (2 * MaxEntries)) + 1 + + Here MaxEntries is the maximum number of entries that the dynamic + table can have. The smallest entry has empty name and value strings + and has the size of 32. Hence, MaxEntries is calculated as: + + MaxEntries = floor( MaxTableCapacity / 32 ) + + MaxTableCapacity is the maximum capacity of the dynamic table as + specified by the decoder; see Section 3.2.3. + + This encoding limits the length of the prefix on long-lived + connections. + + The decoder can reconstruct the Required Insert Count using an + algorithm such as the following. If the decoder encounters a value + of EncodedInsertCount that could not have been produced by a + conformant encoder, it MUST treat this as a connection error of type + QPACK_DECOMPRESSION_FAILED. + + TotalNumberOfInserts is the total number of inserts into the + decoder's dynamic table. + + FullRange = 2 * MaxEntries + if EncodedInsertCount == 0: + ReqInsertCount = 0 + else: + if EncodedInsertCount > FullRange: + Error + MaxValue = TotalNumberOfInserts + MaxEntries + + # MaxWrapped is the largest possible value of + # ReqInsertCount that is 0 mod 2 * MaxEntries + MaxWrapped = floor(MaxValue / FullRange) * FullRange + ReqInsertCount = MaxWrapped + EncodedInsertCount - 1 + + # If ReqInsertCount exceeds MaxValue, the Encoder's value + # must have wrapped one fewer time + if ReqInsertCount > MaxValue: + if ReqInsertCount <= FullRange: + Error + ReqInsertCount -= FullRange + + # Value of 0 must be encoded as 0. + if ReqInsertCount == 0: + Error + + For example, if the dynamic table is 100 bytes, then the Required + Insert Count will be encoded modulo 6. If a decoder has received 10 + inserts, then an encoded value of 4 indicates that the Required + Insert Count is 9 for the field section. + +4.5.1.2. Base + + The Base is used to resolve references in the dynamic table as + described in Section 3.2.5. + + To save space, the Base is encoded relative to the Required Insert + Count using a one-bit Sign ('S' in Figure 12) and the Delta Base + value. A Sign bit of 0 indicates that the Base is greater than or + equal to the value of the Required Insert Count; the decoder adds the + value of Delta Base to the Required Insert Count to determine the + value of the Base. A Sign bit of 1 indicates that the Base is less + than the Required Insert Count; the decoder subtracts the value of + Delta Base from the Required Insert Count and also subtracts one to + determine the value of the Base. That is: + + if Sign == 0: + Base = ReqInsertCount + DeltaBase + else: + Base = ReqInsertCount - DeltaBase - 1 + + A single-pass encoder determines the Base before encoding a field + section. If the encoder inserted entries in the dynamic table while + encoding the field section and is referencing them, Required Insert + Count will be greater than the Base, so the encoded difference is + negative and the Sign bit is set to 1. If the field section was not + encoded using representations that reference the most recent entry in + the table and did not insert any new entries, the Base will be + greater than the Required Insert Count, so the encoded difference + will be positive and the Sign bit is set to 0. + + The value of Base MUST NOT be negative. Though the protocol might + operate correctly with a negative Base using post-Base indexing, it + is unnecessary and inefficient. An endpoint MUST treat a field block + with a Sign bit of 1 as invalid if the value of Required Insert Count + is less than or equal to the value of Delta Base. + + An encoder that produces table updates before encoding a field + section might set Base to the value of Required Insert Count. In + such a case, both the Sign bit and the Delta Base will be set to + zero. + + A field section that was encoded without references to the dynamic + table can use any value for the Base; setting Delta Base to zero is + one of the most efficient encodings. + + For example, with a Required Insert Count of 9, a decoder receives a + Sign bit of 1 and a Delta Base of 2. This sets the Base to 6 and + enables post-Base indexing for three entries. In this example, a + relative index of 1 refers to the fifth entry that was added to the + table; a post-Base index of 1 refers to the eighth entry. + +4.5.2. Indexed Field Line + + An indexed field line representation identifies an entry in the + static table or an entry in the dynamic table with an absolute index + less than the value of the Base. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 1 | T | Index (6+) | + +---+---+-----------------------+ + + Figure 13: Indexed Field Line + + This representation starts with the '1' 1-bit pattern, followed by + the 'T' bit, indicating whether the reference is into the static or + dynamic table. The 6-bit prefix integer (Section 4.1.1) that follows + is used to locate the table entry for the field line. When T=1, the + number represents the static table index; when T=0, the number is the + relative index of the entry in the dynamic table. + +4.5.3. Indexed Field Line with Post-Base Index + + An indexed field line with post-Base index representation identifies + an entry in the dynamic table with an absolute index greater than or + equal to the value of the Base. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 0 | 0 | 0 | 1 | Index (4+) | + +---+---+---+---+---------------+ + + Figure 14: Indexed Field Line with Post-Base Index + + This representation starts with the '0001' 4-bit pattern. This is + followed by the post-Base index (Section 3.2.6) of the matching field + line, represented as an integer with a 4-bit prefix; see + Section 4.1.1. + +4.5.4. Literal Field Line with Name Reference + + A literal field line with name reference representation encodes a + field line where the field name matches the field name of an entry in + the static table or the field name of an entry in the dynamic table + with an absolute index less than the value of the Base. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 0 | 1 | N | T |Name Index (4+)| + +---+---+---+---+---------------+ + | H | Value Length (7+) | + +---+---------------------------+ + | Value String (Length bytes) | + +-------------------------------+ + + Figure 15: Literal Field Line with Name Reference + + This representation starts with the '01' 2-bit pattern. The + following bit, 'N', indicates whether an intermediary is permitted to + add this field line to the dynamic table on subsequent hops. When + the 'N' bit is set, the encoded field line MUST always be encoded + with a literal representation. In particular, when a peer sends a + field line that it received represented as a literal field line with + the 'N' bit set, it MUST use a literal representation to forward this + field line. This bit is intended for protecting field values that + are not to be put at risk by compressing them; see Section 7.1 for + more details. + + The fourth ('T') bit indicates whether the reference is to the static + or dynamic table. The 4-bit prefix integer (Section 4.1.1) that + follows is used to locate the table entry for the field name. When + T=1, the number represents the static table index; when T=0, the + number is the relative index of the entry in the dynamic table. + + Only the field name is taken from the dynamic table entry; the field + value is encoded as an 8-bit prefix string literal; see + Section 4.1.2. + +4.5.5. Literal Field Line with Post-Base Name Reference + + A literal field line with post-Base name reference representation + encodes a field line where the field name matches the field name of a + dynamic table entry with an absolute index greater than or equal to + the value of the Base. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 0 | 0 | 0 | 0 | N |NameIdx(3+)| + +---+---+---+---+---+-----------+ + | H | Value Length (7+) | + +---+---------------------------+ + | Value String (Length bytes) | + +-------------------------------+ + + Figure 16: Literal Field Line with Post-Base Name Reference + + This representation starts with the '0000' 4-bit pattern. The fifth + bit is the 'N' bit as described in Section 4.5.4. This is followed + by a post-Base index of the dynamic table entry (Section 3.2.6) + encoded as an integer with a 3-bit prefix; see Section 4.1.1. + + Only the field name is taken from the dynamic table entry; the field + value is encoded as an 8-bit prefix string literal; see + Section 4.1.2. + +4.5.6. Literal Field Line with Literal Name + + The literal field line with literal name representation encodes a + field name and a field value as string literals. + + 0 1 2 3 4 5 6 7 + +---+---+---+---+---+---+---+---+ + | 0 | 0 | 1 | N | H |NameLen(3+)| + +---+---+---+---+---+-----------+ + | Name String (Length bytes) | + +---+---------------------------+ + | H | Value Length (7+) | + +---+---------------------------+ + | Value String (Length bytes) | + +-------------------------------+ + + Figure 17: Literal Field Line with Literal Name + + This representation starts with the '001' 3-bit pattern. The fourth + bit is the 'N' bit as described in Section 4.5.4. The name follows, + represented as a 4-bit prefix string literal, then the value, + represented as an 8-bit prefix string literal; see Section 4.1.2. + +5. Configuration + + QPACK defines two settings for the HTTP/3 SETTINGS frame: + + SETTINGS_QPACK_MAX_TABLE_CAPACITY (0x01): The default value is zero. + See Section 3.2 for usage. This is the equivalent of the + SETTINGS_HEADER_TABLE_SIZE from HTTP/2. + + SETTINGS_QPACK_BLOCKED_STREAMS (0x07): The default value is zero. + See Section 2.1.2. + +6. Error Handling + + The following error codes are defined for HTTP/3 to indicate failures + of QPACK that prevent the stream or connection from continuing: + + QPACK_DECOMPRESSION_FAILED (0x0200): The decoder failed to interpret + an encoded field section and is not able to continue decoding that + field section. + + QPACK_ENCODER_STREAM_ERROR (0x0201): The decoder failed to interpret + an encoder instruction received on the encoder stream. + + QPACK_DECODER_STREAM_ERROR (0x0202): The encoder failed to interpret + a decoder instruction received on the decoder stream. + +7. Security Considerations + + This section describes potential areas of security concern with + QPACK: + + * Use of compression as a length-based oracle for verifying guesses + about secrets that are compressed into a shared compression + context. + + * Denial of service resulting from exhausting processing or memory + capacity at a decoder. + +7.1. Probing Dynamic Table State + + QPACK reduces the encoded size of field sections by exploiting the + redundancy inherent in protocols like HTTP. The ultimate goal of + this is to reduce the amount of data that is required to send HTTP + requests or responses. + + The compression context used to encode header and trailer fields can + be probed by an attacker who can both define fields to be encoded and + transmitted and observe the length of those fields once they are + encoded. When an attacker can do both, they can adaptively modify + requests in order to confirm guesses about the dynamic table state. + If a guess is compressed into a shorter length, the attacker can + observe the encoded length and infer that the guess was correct. + + This is possible even over the Transport Layer Security Protocol + ([TLS]) and the QUIC Transport Protocol ([QUIC-TRANSPORT]), because + while TLS and QUIC provide confidentiality protection for content, + they only provide a limited amount of protection for the length of + that content. + + | Note: Padding schemes only provide limited protection against + | an attacker with these capabilities, potentially only forcing + | an increased number of guesses to learn the length associated + | with a given guess. Padding schemes also work directly against + | compression by increasing the number of bits that are + | transmitted. + + Attacks like CRIME ([CRIME]) demonstrated the existence of these + general attacker capabilities. The specific attack exploited the + fact that DEFLATE ([RFC1951]) removes redundancy based on prefix + matching. This permitted the attacker to confirm guesses a character + at a time, reducing an exponential-time attack into a linear-time + attack. + +7.1.1. Applicability to QPACK and HTTP + + QPACK mitigates, but does not completely prevent, attacks modeled on + CRIME ([CRIME]) by forcing a guess to match an entire field line + rather than individual characters. An attacker can only learn + whether a guess is correct or not, so the attacker is reduced to a + brute-force guess for the field values associated with a given field + name. + + Therefore, the viability of recovering specific field values depends + on the entropy of values. As a result, values with high entropy are + unlikely to be recovered successfully. However, values with low + entropy remain vulnerable. + + Attacks of this nature are possible any time that two mutually + distrustful entities control requests or responses that are placed + onto a single HTTP/3 connection. If the shared QPACK compressor + permits one entity to add entries to the dynamic table, and the other + to refer to those entries while encoding chosen field lines, then the + attacker (the second entity) can learn the state of the table by + observing the length of the encoded output. + + For example, requests or responses from mutually distrustful entities + can occur when an intermediary either: + + * sends requests from multiple clients on a single connection toward + an origin server, or + + * takes responses from multiple origin servers and places them on a + shared connection toward a client. + + Web browsers also need to assume that requests made on the same + connection by different web origins ([RFC6454]) are made by mutually + distrustful entities. Other scenarios involving mutually distrustful + entities are also possible. + +7.1.2. Mitigation + + Users of HTTP that require confidentiality for header or trailer + fields can use values with entropy sufficient to make guessing + infeasible. However, this is impractical as a general solution + because it forces all users of HTTP to take steps to mitigate + attacks. It would impose new constraints on how HTTP is used. + + Rather than impose constraints on users of HTTP, an implementation of + QPACK can instead constrain how compression is applied in order to + limit the potential for dynamic table probing. + + An ideal solution segregates access to the dynamic table based on the + entity that is constructing the message. Field values that are added + to the table are attributed to an entity, and only the entity that + created a particular value can extract that value. + + To improve compression performance of this option, certain entries + might be tagged as being public. For example, a web browser might + make the values of the Accept-Encoding header field available in all + requests. + + An encoder without good knowledge of the provenance of field values + might instead introduce a penalty for many field lines with the same + field name and different values. This penalty could cause a large + number of attempts to guess a field value to result in the field not + being compared to the dynamic table entries in future messages, + effectively preventing further guesses. + + This response might be made inversely proportional to the length of + the field value. Disabling access to the dynamic table for a given + field name might occur for shorter values more quickly or with higher + probability than for longer values. + + This mitigation is most effective between two endpoints. If messages + are re-encoded by an intermediary without knowledge of which entity + constructed a given message, the intermediary could inadvertently + merge compression contexts that the original encoder had specifically + kept separate. + + | Note: Simply removing entries corresponding to the field from + | the dynamic table can be ineffectual if the attacker has a + | reliable way of causing values to be reinstalled. For example, + | a request to load an image in a web browser typically includes + | the Cookie header field (a potentially highly valued target for + | this sort of attack), and websites can easily force an image to + | be loaded, thereby refreshing the entry in the dynamic table. + +7.1.3. Never-Indexed Literals + + Implementations can also choose to protect sensitive fields by not + compressing them and instead encoding their value as literals. + + Refusing to insert a field line into the dynamic table is only + effective if doing so is avoided on all hops. The never-indexed + literal bit (see Section 4.5.4) can be used to signal to + intermediaries that a particular value was intentionally sent as a + literal. + + An intermediary MUST NOT re-encode a value that uses a literal + representation with the 'N' bit set with another representation that + would index it. If QPACK is used for re-encoding, a literal + representation with the 'N' bit set MUST be used. If HPACK is used + for re-encoding, the never-indexed literal representation (see + Section 6.2.3 of [RFC7541]) MUST be used. + + The choice to mark that a field value should never be indexed depends + on several factors. Since QPACK does not protect against guessing an + entire field value, short or low-entropy values are more readily + recovered by an adversary. Therefore, an encoder might choose not to + index values with low entropy. + + An encoder might also choose not to index values for fields that are + considered to be highly valuable or sensitive to recovery, such as + the Cookie or Authorization header fields. + + On the contrary, an encoder might prefer indexing values for fields + that have little or no value if they were exposed. For instance, a + User-Agent header field does not commonly vary between requests and + is sent to any server. In that case, confirmation that a particular + User-Agent value has been used provides little value. + + Note that these criteria for deciding to use a never-indexed literal + representation will evolve over time as new attacks are discovered. + +7.2. Static Huffman Encoding + + There is no currently known attack against a static Huffman encoding. + A study has shown that using a static Huffman encoding table created + an information leakage; however, this same study concluded that an + attacker could not take advantage of this information leakage to + recover any meaningful amount of information (see [PETAL]). + +7.3. Memory Consumption + + An attacker can try to cause an endpoint to exhaust its memory. + QPACK is designed to limit both the peak and stable amounts of memory + allocated by an endpoint. + + QPACK uses the definition of the maximum size of the dynamic table + and the maximum number of blocking streams to limit the amount of + memory the encoder can cause the decoder to consume. In HTTP/3, + these values are controlled by the decoder through the settings + parameters SETTINGS_QPACK_MAX_TABLE_CAPACITY and + SETTINGS_QPACK_BLOCKED_STREAMS, respectively (see Section 3.2.3 and + Section 2.1.2). The limit on the size of the dynamic table takes + into account the size of the data stored in the dynamic table, plus a + small allowance for overhead. The limit on the number of blocked + streams is only a proxy for the maximum amount of memory required by + the decoder. The actual maximum amount of memory will depend on how + much memory the decoder uses to track each blocked stream. + + A decoder can limit the amount of state memory used for the dynamic + table by setting an appropriate value for the maximum size of the + dynamic table. In HTTP/3, this is realized by setting an appropriate + value for the SETTINGS_QPACK_MAX_TABLE_CAPACITY parameter. An + encoder can limit the amount of state memory it uses by choosing a + smaller dynamic table size than the decoder allows and signaling this + to the decoder (see Section 4.3.1). + + A decoder can limit the amount of state memory used for blocked + streams by setting an appropriate value for the maximum number of + blocked streams. In HTTP/3, this is realized by setting an + appropriate value for the SETTINGS_QPACK_BLOCKED_STREAMS parameter. + Streams that risk becoming blocked consume no additional state memory + on the encoder. + + An encoder allocates memory to track all dynamic table references in + unacknowledged field sections. An implementation can directly limit + the amount of state memory by only using as many references to the + dynamic table as it wishes to track; no signaling to the decoder is + required. However, limiting references to the dynamic table will + reduce compression effectiveness. + + The amount of temporary memory consumed by an encoder or decoder can + be limited by processing field lines sequentially. A decoder + implementation does not need to retain a complete list of field lines + while decoding a field section. An encoder implementation does not + need to retain a complete list of field lines while encoding a field + section if it is using a single-pass algorithm. Note that it might + be necessary for an application to retain a complete list of field + lines for other reasons; even if QPACK does not force this to occur, + application constraints might make this necessary. + + While the negotiated limit on the dynamic table size accounts for + much of the memory that can be consumed by a QPACK implementation, + data that cannot be immediately sent due to flow control is not + affected by this limit. Implementations should limit the size of + unsent data, especially on the decoder stream where flexibility to + choose what to send is limited. Possible responses to an excess of + unsent data might include limiting the ability of the peer to open + new streams, reading only from the encoder stream, or closing the + connection. + +7.4. Implementation Limits + + An implementation of QPACK needs to ensure that large values for + integers, long encoding for integers, or long string literals do not + create security weaknesses. + + An implementation has to set a limit for the values it accepts for + integers, as well as for the encoded length; see Section 4.1.1. In + the same way, it has to set a limit to the length it accepts for + string literals; see Section 4.1.2. These limits SHOULD be large + enough to process the largest individual field the HTTP + implementation can be configured to accept. + + If an implementation encounters a value larger than it is able to + decode, this MUST be treated as a stream error of type + QPACK_DECOMPRESSION_FAILED if on a request stream or a connection + error of the appropriate type if on the encoder or decoder stream. + +8. IANA Considerations + + This document makes multiple registrations in the registries defined + by [HTTP/3]. The allocations created by this document are all + assigned permanent status and list a change controller of the IETF + and a contact of the HTTP working group (ietf-http-wg@w3.org). + +8.1. Settings Registration + + This document specifies two settings. The entries in the following + table are registered in the "HTTP/3 Settings" registry established in + [HTTP/3]. + + +==========================+======+===============+=========+ + | Setting Name | Code | Specification | Default | + +==========================+======+===============+=========+ + | QPACK_MAX_TABLE_CAPACITY | 0x01 | Section 5 | 0 | + +--------------------------+------+---------------+---------+ + | QPACK_BLOCKED_STREAMS | 0x07 | Section 5 | 0 | + +--------------------------+------+---------------+---------+ + + Table 1: Additions to the HTTP/3 Settings Registry + + For formatting reasons, the setting names here are abbreviated by + removing the 'SETTINGS_' prefix. + +8.2. Stream Type Registration + + This document specifies two stream types. The entries in the + following table are registered in the "HTTP/3 Stream Types" registry + established in [HTTP/3]. + + +======================+======+===============+========+ + | Stream Type | Code | Specification | Sender | + +======================+======+===============+========+ + | QPACK Encoder Stream | 0x02 | Section 4.2 | Both | + +----------------------+------+---------------+--------+ + | QPACK Decoder Stream | 0x03 | Section 4.2 | Both | + +----------------------+------+---------------+--------+ + + Table 2: Additions to the HTTP/3 Stream Types Registry + +8.3. Error Code Registration + + This document specifies three error codes. The entries in the + following table are registered in the "HTTP/3 Error Codes" registry + established in [HTTP/3]. + + +============================+========+=============+===============+ + | Name | Code |Description | Specification | + +============================+========+=============+===============+ + | QPACK_DECOMPRESSION_FAILED | 0x0200 |Decoding of a| Section 6 | + | | |field section| | + | | |failed | | + +----------------------------+--------+-------------+---------------+ + | QPACK_ENCODER_STREAM_ERROR | 0x0201 |Error on the | Section 6 | + | | |encoder | | + | | |stream | | + +----------------------------+--------+-------------+---------------+ + | QPACK_DECODER_STREAM_ERROR | 0x0202 |Error on the | Section 6 | + | | |decoder | | + | | |stream | | + +----------------------------+--------+-------------+---------------+ + + Table 3: Additions to the HTTP/3 Error Codes Registry + +9. References + +9.1. Normative References + + [HTTP] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP Semantics", STD 97, RFC 9110, + DOI 10.17487/RFC9110, June 2022, + <https://www.rfc-editor.org/info/rfc9110>. + + [HTTP/3] Bishop, M., Ed., "HTTP/3", RFC 9114, DOI 10.17487/RFC9114, + June 2022, <https://www.rfc-editor.org/info/rfc9114>. + + [QUIC-TRANSPORT] + Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC2360] Scott, G., "Guide for Internet Standards Writers", BCP 22, + RFC 2360, DOI 10.17487/RFC2360, June 1998, + <https://www.rfc-editor.org/info/rfc2360>. + + [RFC7541] Peon, R. and H. Ruellan, "HPACK: Header Compression for + HTTP/2", RFC 7541, DOI 10.17487/RFC7541, May 2015, + <https://www.rfc-editor.org/info/rfc7541>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + +9.2. Informative References + + [CRIME] Wikipedia, "CRIME", May 2015, <http://en.wikipedia.org/w/ + index.php?title=CRIME&oldid=660948120>. + + [HTTP/2] Thomson, M., Ed. and C. Benfield, Ed., "HTTP/2", RFC 9113, + DOI 10.17487/RFC9113, June 2022, + <https://www.rfc-editor.org/info/rfc9113>. + + [PETAL] Tan, J. and J. Nahata, "PETAL: Preset Encoding + Table Information Leakage", April 2013, + <http://www.pdl.cmu.edu/PDL-FTP/associated/CMU-PDL- + 13-106.pdf>. + + [RFC1951] Deutsch, P., "DEFLATE Compressed Data Format Specification + version 1.3", RFC 1951, DOI 10.17487/RFC1951, May 1996, + <https://www.rfc-editor.org/info/rfc1951>. + + [RFC6454] Barth, A., "The Web Origin Concept", RFC 6454, + DOI 10.17487/RFC6454, December 2011, + <https://www.rfc-editor.org/info/rfc6454>. + + [TLS] Rescorla, E., "The Transport Layer Security (TLS) Protocol + Version 1.3", RFC 8446, DOI 10.17487/RFC8446, August 2018, + <https://www.rfc-editor.org/info/rfc8446>. + +Appendix A. Static Table + + This table was generated by analyzing actual Internet traffic in 2018 + and including the most common header fields, after filtering out some + unsupported and non-standard values. Due to this methodology, some + of the entries may be inconsistent or appear multiple times with + similar but not identical values. The order of the entries is + optimized to encode the most common header fields with the smallest + number of bytes. + + +=======+==================================+=======================+ + | Index | Name | Value | + +=======+==================================+=======================+ + | 0 | :authority | | + +-------+----------------------------------+-----------------------+ + | 1 | :path | / | + +-------+----------------------------------+-----------------------+ + | 2 | age | 0 | + +-------+----------------------------------+-----------------------+ + | 3 | content-disposition | | + +-------+----------------------------------+-----------------------+ + | 4 | content-length | 0 | + +-------+----------------------------------+-----------------------+ + | 5 | cookie | | + +-------+----------------------------------+-----------------------+ + | 6 | date | | + +-------+----------------------------------+-----------------------+ + | 7 | etag | | + +-------+----------------------------------+-----------------------+ + | 8 | if-modified-since | | + +-------+----------------------------------+-----------------------+ + | 9 | if-none-match | | + +-------+----------------------------------+-----------------------+ + | 10 | last-modified | | + +-------+----------------------------------+-----------------------+ + | 11 | link | | + +-------+----------------------------------+-----------------------+ + | 12 | location | | + +-------+----------------------------------+-----------------------+ + | 13 | referer | | + +-------+----------------------------------+-----------------------+ + | 14 | set-cookie | | + +-------+----------------------------------+-----------------------+ + | 15 | :method | CONNECT | + +-------+----------------------------------+-----------------------+ + | 16 | :method | DELETE | + +-------+----------------------------------+-----------------------+ + | 17 | :method | GET | + +-------+----------------------------------+-----------------------+ + | 18 | :method | HEAD | + +-------+----------------------------------+-----------------------+ + | 19 | :method | OPTIONS | + +-------+----------------------------------+-----------------------+ + | 20 | :method | POST | + +-------+----------------------------------+-----------------------+ + | 21 | :method | PUT | + +-------+----------------------------------+-----------------------+ + | 22 | :scheme | http | + +-------+----------------------------------+-----------------------+ + | 23 | :scheme | https | + +-------+----------------------------------+-----------------------+ + | 24 | :status | 103 | + +-------+----------------------------------+-----------------------+ + | 25 | :status | 200 | + +-------+----------------------------------+-----------------------+ + | 26 | :status | 304 | + +-------+----------------------------------+-----------------------+ + | 27 | :status | 404 | + +-------+----------------------------------+-----------------------+ + | 28 | :status | 503 | + +-------+----------------------------------+-----------------------+ + | 29 | accept | */* | + +-------+----------------------------------+-----------------------+ + | 30 | accept | application/dns- | + | | | message | + +-------+----------------------------------+-----------------------+ + | 31 | accept-encoding | gzip, deflate, br | + +-------+----------------------------------+-----------------------+ + | 32 | accept-ranges | bytes | + +-------+----------------------------------+-----------------------+ + | 33 | access-control-allow-headers | cache-control | + +-------+----------------------------------+-----------------------+ + | 34 | access-control-allow-headers | content-type | + +-------+----------------------------------+-----------------------+ + | 35 | access-control-allow-origin | * | + +-------+----------------------------------+-----------------------+ + | 36 | cache-control | max-age=0 | + +-------+----------------------------------+-----------------------+ + | 37 | cache-control | max-age=2592000 | + +-------+----------------------------------+-----------------------+ + | 38 | cache-control | max-age=604800 | + +-------+----------------------------------+-----------------------+ + | 39 | cache-control | no-cache | + +-------+----------------------------------+-----------------------+ + | 40 | cache-control | no-store | + +-------+----------------------------------+-----------------------+ + | 41 | cache-control | public, max- | + | | | age=31536000 | + +-------+----------------------------------+-----------------------+ + | 42 | content-encoding | br | + +-------+----------------------------------+-----------------------+ + | 43 | content-encoding | gzip | + +-------+----------------------------------+-----------------------+ + | 44 | content-type | application/dns- | + | | | message | + +-------+----------------------------------+-----------------------+ + | 45 | content-type | application/ | + | | | javascript | + +-------+----------------------------------+-----------------------+ + | 46 | content-type | application/json | + +-------+----------------------------------+-----------------------+ + | 47 | content-type | application/x-www- | + | | | form-urlencoded | + +-------+----------------------------------+-----------------------+ + | 48 | content-type | image/gif | + +-------+----------------------------------+-----------------------+ + | 49 | content-type | image/jpeg | + +-------+----------------------------------+-----------------------+ + | 50 | content-type | image/png | + +-------+----------------------------------+-----------------------+ + | 51 | content-type | text/css | + +-------+----------------------------------+-----------------------+ + | 52 | content-type | text/html; | + | | | charset=utf-8 | + +-------+----------------------------------+-----------------------+ + | 53 | content-type | text/plain | + +-------+----------------------------------+-----------------------+ + | 54 | content-type | text/ | + | | | plain;charset=utf-8 | + +-------+----------------------------------+-----------------------+ + | 55 | range | bytes=0- | + +-------+----------------------------------+-----------------------+ + | 56 | strict-transport-security | max-age=31536000 | + +-------+----------------------------------+-----------------------+ + | 57 | strict-transport-security | max-age=31536000; | + | | | includesubdomains | + +-------+----------------------------------+-----------------------+ + | 58 | strict-transport-security | max-age=31536000; | + | | | includesubdomains; | + | | | preload | + +-------+----------------------------------+-----------------------+ + | 59 | vary | accept-encoding | + +-------+----------------------------------+-----------------------+ + | 60 | vary | origin | + +-------+----------------------------------+-----------------------+ + | 61 | x-content-type-options | nosniff | + +-------+----------------------------------+-----------------------+ + | 62 | x-xss-protection | 1; mode=block | + +-------+----------------------------------+-----------------------+ + | 63 | :status | 100 | + +-------+----------------------------------+-----------------------+ + | 64 | :status | 204 | + +-------+----------------------------------+-----------------------+ + | 65 | :status | 206 | + +-------+----------------------------------+-----------------------+ + | 66 | :status | 302 | + +-------+----------------------------------+-----------------------+ + | 67 | :status | 400 | + +-------+----------------------------------+-----------------------+ + | 68 | :status | 403 | + +-------+----------------------------------+-----------------------+ + | 69 | :status | 421 | + +-------+----------------------------------+-----------------------+ + | 70 | :status | 425 | + +-------+----------------------------------+-----------------------+ + | 71 | :status | 500 | + +-------+----------------------------------+-----------------------+ + | 72 | accept-language | | + +-------+----------------------------------+-----------------------+ + | 73 | access-control-allow-credentials | FALSE | + +-------+----------------------------------+-----------------------+ + | 74 | access-control-allow-credentials | TRUE | + +-------+----------------------------------+-----------------------+ + | 75 | access-control-allow-headers | * | + +-------+----------------------------------+-----------------------+ + | 76 | access-control-allow-methods | get | + +-------+----------------------------------+-----------------------+ + | 77 | access-control-allow-methods | get, post, options | + +-------+----------------------------------+-----------------------+ + | 78 | access-control-allow-methods | options | + +-------+----------------------------------+-----------------------+ + | 79 | access-control-expose-headers | content-length | + +-------+----------------------------------+-----------------------+ + | 80 | access-control-request-headers | content-type | + +-------+----------------------------------+-----------------------+ + | 81 | access-control-request-method | get | + +-------+----------------------------------+-----------------------+ + | 82 | access-control-request-method | post | + +-------+----------------------------------+-----------------------+ + | 83 | alt-svc | clear | + +-------+----------------------------------+-----------------------+ + | 84 | authorization | | + +-------+----------------------------------+-----------------------+ + | 85 | content-security-policy | script-src 'none'; | + | | | object-src 'none'; | + | | | base-uri 'none' | + +-------+----------------------------------+-----------------------+ + | 86 | early-data | 1 | + +-------+----------------------------------+-----------------------+ + | 87 | expect-ct | | + +-------+----------------------------------+-----------------------+ + | 88 | forwarded | | + +-------+----------------------------------+-----------------------+ + | 89 | if-range | | + +-------+----------------------------------+-----------------------+ + | 90 | origin | | + +-------+----------------------------------+-----------------------+ + | 91 | purpose | prefetch | + +-------+----------------------------------+-----------------------+ + | 92 | server | | + +-------+----------------------------------+-----------------------+ + | 93 | timing-allow-origin | * | + +-------+----------------------------------+-----------------------+ + | 94 | upgrade-insecure-requests | 1 | + +-------+----------------------------------+-----------------------+ + | 95 | user-agent | | + +-------+----------------------------------+-----------------------+ + | 96 | x-forwarded-for | | + +-------+----------------------------------+-----------------------+ + | 97 | x-frame-options | deny | + +-------+----------------------------------+-----------------------+ + | 98 | x-frame-options | sameorigin | + +-------+----------------------------------+-----------------------+ + + Table 4: Static Table + + Any line breaks that appear within field names or values are due to + formatting. + +Appendix B. Encoding and Decoding Examples + + The following examples represent a series of exchanges between an + encoder and a decoder. The exchanges are designed to exercise most + QPACK instructions and highlight potentially common patterns and + their impact on dynamic table state. The encoder sends three encoded + field sections containing one field line each, as well as two + speculative inserts that are not referenced. + + The state of the encoder's dynamic table is shown, along with its + current size. Each entry is shown with the Absolute Index of the + entry (Abs), the current number of outstanding encoded field sections + with references to that entry (Ref), along with the name and value. + Entries above the 'acknowledged' line have been acknowledged by the + decoder. + +B.1. Literal Field Line with Name Reference + + The encoder sends an encoded field section containing a literal + representation of a field with a static name reference. + + Data | Interpretation + | Encoder's Dynamic Table + + Stream: 0 + 0000 | Required Insert Count = 0, Base = 0 + 510b 2f69 6e64 6578 | Literal Field Line with Name Reference + 2e68 746d 6c | Static Table, Index=1 + | (:path=/index.html) + + Abs Ref Name Value + ^-- acknowledged --^ + Size=0 + +B.2. Dynamic Table + + The encoder sets the dynamic table capacity, inserts a header with a + dynamic name reference, then sends a potentially blocking, encoded + field section referencing this new entry. The decoder acknowledges + processing the encoded field section, which implicitly acknowledges + all dynamic table insertions up to the Required Insert Count. + + Stream: Encoder + 3fbd01 | Set Dynamic Table Capacity=220 + c00f 7777 772e 6578 | Insert With Name Reference + 616d 706c 652e 636f | Static Table, Index=0 + 6d | (:authority=www.example.com) + c10c 2f73 616d 706c | Insert With Name Reference + 652f 7061 7468 | Static Table, Index=1 + | (:path=/sample/path) + + Abs Ref Name Value + ^-- acknowledged --^ + 0 0 :authority www.example.com + 1 0 :path /sample/path + Size=106 + + Stream: 4 + 0381 | Required Insert Count = 2, Base = 0 + 10 | Indexed Field Line With Post-Base Index + | Absolute Index = Base(0) + Index(0) = 0 + | (:authority=www.example.com) + 11 | Indexed Field Line With Post-Base Index + | Absolute Index = Base(0) + Index(1) = 1 + | (:path=/sample/path) + + Abs Ref Name Value + ^-- acknowledged --^ + 0 1 :authority www.example.com + 1 1 :path /sample/path + Size=106 + + Stream: Decoder + 84 | Section Acknowledgment (stream=4) + + Abs Ref Name Value + 0 0 :authority www.example.com + 1 0 :path /sample/path + ^-- acknowledged --^ + Size=106 + +B.3. Speculative Insert + + The encoder inserts a header into the dynamic table with a literal + name. The decoder acknowledges receipt of the entry. The encoder + does not send any encoded field sections. + + Stream: Encoder + 4a63 7573 746f 6d2d | Insert With Literal Name + 6b65 790c 6375 7374 | (custom-key=custom-value) + 6f6d 2d76 616c 7565 | + + Abs Ref Name Value + 0 0 :authority www.example.com + 1 0 :path /sample/path + ^-- acknowledged --^ + 2 0 custom-key custom-value + Size=160 + + Stream: Decoder + 01 | Insert Count Increment (1) + + Abs Ref Name Value + 0 0 :authority www.example.com + 1 0 :path /sample/path + 2 0 custom-key custom-value + ^-- acknowledged --^ + Size=160 + +B.4. Duplicate Instruction, Stream Cancellation + + The encoder duplicates an existing entry in the dynamic table, then + sends an encoded field section referencing the dynamic table entries + including the duplicated entry. The packet containing the encoder + stream data is delayed. Before the packet arrives, the decoder + cancels the stream and notifies the encoder that the encoded field + section was not processed. + + Stream: Encoder + 02 | Duplicate (Relative Index = 2) + | Absolute Index = + | Insert Count(3) - Index(2) - 1 = 0 + + Abs Ref Name Value + 0 0 :authority www.example.com + 1 0 :path /sample/path + 2 0 custom-key custom-value + ^-- acknowledged --^ + 3 0 :authority www.example.com + Size=217 + + Stream: 8 + 0500 | Required Insert Count = 4, Base = 4 + 80 | Indexed Field Line, Dynamic Table + | Absolute Index = Base(4) - Index(0) - 1 = 3 + | (:authority=www.example.com) + c1 | Indexed Field Line, Static Table Index = 1 + | (:path=/) + 81 | Indexed Field Line, Dynamic Table + | Absolute Index = Base(4) - Index(1) - 1 = 2 + | (custom-key=custom-value) + + Abs Ref Name Value + 0 0 :authority www.example.com + 1 0 :path /sample/path + 2 1 custom-key custom-value + ^-- acknowledged --^ + 3 1 :authority www.example.com + Size=217 + + Stream: Decoder + 48 | Stream Cancellation (Stream=8) + + Abs Ref Name Value + 0 0 :authority www.example.com + 1 0 :path /sample/path + 2 0 custom-key custom-value + ^-- acknowledged --^ + 3 0 :authority www.example.com + Size=217 + +B.5. Dynamic Table Insert, Eviction + + The encoder inserts another header into the dynamic table, which + evicts the oldest entry. The encoder does not send any encoded field + sections. + + Stream: Encoder + 810d 6375 7374 6f6d | Insert With Name Reference + 2d76 616c 7565 32 | Dynamic Table, Relative Index = 1 + | Absolute Index = + | Insert Count(4) - Index(1) - 1 = 2 + | (custom-key=custom-value2) + + Abs Ref Name Value + 1 0 :path /sample/path + 2 0 custom-key custom-value + ^-- acknowledged --^ + 3 0 :authority www.example.com + 4 0 custom-key custom-value2 + Size=215 + +Appendix C. Sample Single-Pass Encoding Algorithm + + Pseudocode for single-pass encoding, excluding handling of + duplicates, non-blocking mode, available encoder stream flow control + and reference tracking. + + # Helper functions: + # ==== + # Encode an integer with the specified prefix and length + encodeInteger(buffer, prefix, value, prefixLength) + + # Encode a dynamic table insert instruction with optional static + # or dynamic name index (but not both) + encodeInsert(buffer, staticNameIndex, dynamicNameIndex, fieldLine) + + # Encode a static index reference + encodeStaticIndexReference(buffer, staticIndex) + + # Encode a dynamic index reference relative to Base + encodeDynamicIndexReference(buffer, dynamicIndex, base) + + # Encode a literal with an optional static name index + encodeLiteral(buffer, staticNameIndex, fieldLine) + + # Encode a literal with a dynamic name index relative to Base + encodeDynamicLiteral(buffer, dynamicNameIndex, base, fieldLine) + + # Encoding Algorithm + # ==== + base = dynamicTable.getInsertCount() + requiredInsertCount = 0 + for line in fieldLines: + staticIndex = staticTable.findIndex(line) + if staticIndex is not None: + encodeStaticIndexReference(streamBuffer, staticIndex) + continue + + dynamicIndex = dynamicTable.findIndex(line) + if dynamicIndex is None: + # No matching entry. Either insert+index or encode literal + staticNameIndex = staticTable.findName(line.name) + if staticNameIndex is None: + dynamicNameIndex = dynamicTable.findName(line.name) + + if shouldIndex(line) and dynamicTable.canIndex(line): + encodeInsert(encoderBuffer, staticNameIndex, + dynamicNameIndex, line) + dynamicIndex = dynamicTable.add(line) + + if dynamicIndex is None: + # Could not index it, literal + if dynamicNameIndex is not None: + # Encode literal with dynamic name, possibly above Base + encodeDynamicLiteral(streamBuffer, dynamicNameIndex, + base, line) + requiredInsertCount = max(requiredInsertCount, + dynamicNameIndex) + else: + # Encodes a literal with a static name or literal name + encodeLiteral(streamBuffer, staticNameIndex, line) + else: + # Dynamic index reference + assert(dynamicIndex is not None) + requiredInsertCount = max(requiredInsertCount, dynamicIndex) + # Encode dynamicIndex, possibly above Base + encodeDynamicIndexReference(streamBuffer, dynamicIndex, base) + + # encode the prefix + if requiredInsertCount == 0: + encodeInteger(prefixBuffer, 0x00, 0, 8) + encodeInteger(prefixBuffer, 0x00, 0, 7) + else: + wireRIC = ( + requiredInsertCount + % (2 * getMaxEntries(maxTableCapacity)) + ) + 1; + encodeInteger(prefixBuffer, 0x00, wireRIC, 8) + if base >= requiredInsertCount: + encodeInteger(prefixBuffer, 0x00, + base - requiredInsertCount, 7) + else: + encodeInteger(prefixBuffer, 0x80, + requiredInsertCount - base - 1, 7) + + return encoderBuffer, prefixBuffer + streamBuffer + +Acknowledgments + + The IETF QUIC Working Group received an enormous amount of support + from many people. + + The compression design team did substantial work exploring the + problem space and influencing the initial draft version of this + document. The contributions of design team members Roberto Peon, + Martin Thomson, and Dmitri Tikhonov are gratefully acknowledged. + + The following people also provided substantial contributions to this + document: + + * Bence Beky + * Alessandro Ghedini + * Ryan Hamilton + * Robin Marx + * Patrick McManus + * ๅฅฅ ไธ€็ฉ‚ (Kazuho Oku) + * Lucas Pardue + * Biren Roy + * Ian Swett + + This document draws heavily on the text of [RFC7541]. The indirect + input of those authors is also gratefully acknowledged. + + Buck Krasic's contribution was supported by Google during his + employment there. + + A portion of Mike Bishop's contribution was supported by Microsoft + during his employment there. + +Authors' Addresses + + Charles 'Buck' Krasic + Email: krasic@acm.org + + + Mike Bishop + Akamai Technologies + Email: mbishop@evequefou.be + + + Alan Frindell (editor) + Facebook + Email: afrind@fb.com diff --git a/eval/corpora/rfc/RFC 9218 - Extensible Prioritization Scheme for HTTP.txt b/eval/corpora/rfc/RFC 9218 - Extensible Prioritization Scheme for HTTP.txt new file mode 100644 index 00000000..64254179 --- /dev/null +++ b/eval/corpora/rfc/RFC 9218 - Extensible Prioritization Scheme for HTTP.txt @@ -0,0 +1,1149 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) ๅฅฅ ไธ€็ฉ‚ (K. Oku) +Request for Comments: 9218 Fastly +Category: Standards Track L. Pardue +ISSN: 2070-1721 Cloudflare + June 2022 + + + Extensible Prioritization Scheme for HTTP + +Abstract + + This document describes a scheme that allows an HTTP client to + communicate its preferences for how the upstream server prioritizes + responses to its requests, and also allows a server to hint to a + downstream intermediary how its responses should be prioritized when + they are forwarded. This document defines the Priority header field + for communicating the initial priority in an HTTP version-independent + manner, as well as HTTP/2 and HTTP/3 frames for reprioritizing + responses. These share a common format structure that is designed to + provide future extensibility. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9218. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 1.1. Notational Conventions + 2. Motivation for Replacing RFC 7540 Stream Priorities + 2.1. Disabling RFC 7540 Stream Priorities + 2.1.1. Advice when Using Extensible Priorities as the + Alternative + 3. Applicability of the Extensible Priority Scheme + 4. Priority Parameters + 4.1. Urgency + 4.2. Incremental + 4.3. Defining New Priority Parameters + 4.3.1. Registration + 5. The Priority HTTP Header Field + 6. Reprioritization + 7. The PRIORITY_UPDATE Frame + 7.1. HTTP/2 PRIORITY_UPDATE Frame + 7.2. HTTP/3 PRIORITY_UPDATE Frame + 8. Merging Client- and Server-Driven Priority Parameters + 9. Client Scheduling + 10. Server Scheduling + 10.1. Intermediaries with Multiple Backend Connections + 11. Scheduling and the CONNECT Method + 12. Retransmission Scheduling + 13. Fairness + 13.1. Coalescing Intermediaries + 13.2. HTTP/1.x Back Ends + 13.3. Intentional Introduction of Unfairness + 14. Why Use an End-to-End Header Field? + 15. Security Considerations + 16. IANA Considerations + 17. References + 17.1. Normative References + 17.2. Informative References + Acknowledgements + Authors' Addresses + +1. Introduction + + It is common for representations of an HTTP [HTTP] resource to have + relationships to one or more other resources. Clients will often + discover these relationships while processing a retrieved + representation, which may lead to further retrieval requests. + Meanwhile, the nature of the relationships determines whether a + client is blocked from continuing to process locally available + resources. An example of this is the visual rendering of an HTML + document, which could be blocked by the retrieval of a Cascading + Style Sheets (CSS) file that the document refers to. In contrast, + inline images do not block rendering and get drawn incrementally as + the chunks of the images arrive. + + HTTP/2 [HTTP/2] and HTTP/3 [HTTP/3] support multiplexing of requests + and responses in a single connection. An important feature of any + implementation of a protocol that provides multiplexing is the + ability to prioritize the sending of information. For example, to + provide meaningful presentation of an HTML document at the earliest + moment, it is important for an HTTP server to prioritize the HTTP + responses, or the chunks of those HTTP responses, that it sends to a + client. + + HTTP/2 and HTTP/3 servers can schedule transmission of concurrent + response data by any means they choose. Servers can ignore client + priority signals and still successfully serve HTTP responses. + However, servers that operate in ignorance of how clients issue + requests and consume responses can cause suboptimal client + application performance. Priority signals allow clients to + communicate their view of request priority. Servers have their own + needs that are independent of client needs, so they often combine + priority signals with other available information in order to inform + scheduling of response data. + + RFC 7540 [RFC7540] stream priority allowed a client to send a series + of priority signals that communicate to the server a "priority tree"; + the structure of this tree represents the client's preferred relative + ordering and weighted distribution of the bandwidth among HTTP + responses. Servers could use these priority signals as input into + prioritization decisions. + + The design and implementation of RFC 7540 stream priority were + observed to have shortcomings, as explained in Section 2. HTTP/2 + [HTTP/2] has consequently deprecated the use of these stream priority + signals. The prioritization scheme and priority signals defined + herein can act as a substitute for RFC 7540 stream priority. + + This document describes an extensible scheme for prioritizing HTTP + responses that uses absolute values. Section 4 defines priority + parameters, which are a standardized and extensible format of + priority information. Section 5 defines the Priority HTTP header + field, which is an end-to-end priority signal that is independent of + protocol version. Clients can send this header field to signal their + view of how responses should be prioritized. Similarly, servers + behind an intermediary can use it to signal priority to the + intermediary. After sending a request, a client can change their + view of response priority (see Section 6) by sending HTTP-version- + specific frames as defined in Sections 7.1 and 7.2. + + Header field and frame priority signals are input to a server's + response prioritization process. They are only a suggestion and do + not guarantee any particular processing or transmission order for one + response relative to any other response. Sections 10 and 12 provide + considerations and guidance about how servers might act upon signals. + +1.1. Notational Conventions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + This document uses the following terminology from Section 3 of + [STRUCTURED-FIELDS] to specify syntax and parsing: "Boolean", + "Dictionary", and "Integer". + + Example HTTP requests and responses use the HTTP/2-style formatting + from [HTTP/2]. + + This document uses the variable-length integer encoding from [QUIC]. + + The term "control stream" is used to describe both the HTTP/2 stream + with identifier 0x0 and the HTTP/3 control stream; see Section 6.2.1 + of [HTTP/3]. + + The term "HTTP/2 priority signal" is used to describe the priority + information sent from clients to servers in HTTP/2 frames; see + Section 5.3.2 of [HTTP/2]. + +2. Motivation for Replacing RFC 7540 Stream Priorities + + RFC 7540 stream priority (see Section 5.3 of [RFC7540]) is a complex + system where clients signal stream dependencies and weights to + describe an unbalanced tree. It suffered from limited deployment and + interoperability and has been deprecated in a revision of HTTP/2 + [HTTP/2]. HTTP/2 retains these protocol elements in order to + maintain wire compatibility (see Section 5.3.2 of [HTTP/2]), which + means that they might still be used even in the presence of + alternative signaling, such as the scheme this document describes. + + Many RFC 7540 server implementations do not act on HTTP/2 priority + signals. + + Prioritization can use information that servers have about resources + or the order in which requests are generated. For example, a server, + with knowledge of an HTML document structure, might want to + prioritize the delivery of images that are critical to user + experience above other images. With RFC 7540, it is difficult for + servers to interpret signals from clients for prioritization, as the + same conditions could result in very different signaling from + different clients. This document describes signaling that is simpler + and more constrained, requiring less interpretation and allowing less + variation. + + RFC 7540 does not define a method that can be used by a server to + provide a priority signal for intermediaries. + + RFC 7540 stream priority is expressed relative to other requests + sharing the same connection at the same time. It is difficult to + incorporate such a design into applications that generate requests + without knowledge of how other requests might share a connection, or + into protocols that do not have strong ordering guarantees across + streams, like HTTP/3 [HTTP/3]. + + Experiments from independent research [MARX] have shown that simpler + schemes can reach at least equivalent performance characteristics + compared to the more complex RFC 7540 setups seen in practice, at + least for the Web use case. + +2.1. Disabling RFC 7540 Stream Priorities + + The problems and insights set out above provided the motivation for + an alternative to RFC 7540 stream priority (see Section 5.3 of + [HTTP/2]). + + The SETTINGS_NO_RFC7540_PRIORITIES HTTP/2 setting is defined by this + document in order to allow endpoints to omit or ignore HTTP/2 + priority signals (see Section 5.3.2 of [HTTP/2]), as described below. + The value of SETTINGS_NO_RFC7540_PRIORITIES MUST be 0 or 1. Any + value other than 0 or 1 MUST be treated as a connection error (see + Section 5.4.1 of [HTTP/2]) of type PROTOCOL_ERROR. The initial value + is 0. + + If endpoints use SETTINGS_NO_RFC7540_PRIORITIES, they MUST send it in + the first SETTINGS frame. Senders MUST NOT change the + SETTINGS_NO_RFC7540_PRIORITIES value after the first SETTINGS frame. + Receivers that detect a change MAY treat it as a connection error of + type PROTOCOL_ERROR. + + Clients can send SETTINGS_NO_RFC7540_PRIORITIES with a value of 1 to + indicate that they are not using HTTP/2 priority signals. The + SETTINGS frame precedes any HTTP/2 priority signal sent from clients, + so servers can determine whether they need to allocate any resources + to signal handling before signals arrive. A server that receives + SETTINGS_NO_RFC7540_PRIORITIES with a value of 1 MUST ignore HTTP/2 + priority signals. + + Servers can send SETTINGS_NO_RFC7540_PRIORITIES with a value of 1 to + indicate that they will ignore HTTP/2 priority signals sent by + clients. + + Endpoints that send SETTINGS_NO_RFC7540_PRIORITIES are encouraged to + use alternative priority signals (for example, see Section 5 or + Section 7.1), but there is no requirement to use a specific signal + type. + +2.1.1. Advice when Using Extensible Priorities as the Alternative + + Before receiving a SETTINGS frame from a server, a client does not + know if the server is ignoring HTTP/2 priority signals. Therefore, + until the client receives the SETTINGS frame from the server, the + client SHOULD send both the HTTP/2 priority signals and the signals + of this prioritization scheme (see Sections 5 and 7.1). + + Once the client receives the first SETTINGS frame that contains the + SETTINGS_NO_RFC7540_PRIORITIES parameter with a value of 1, it SHOULD + stop sending the HTTP/2 priority signals. This avoids sending + redundant signals that are known to be ignored. + + Similarly, if the client receives SETTINGS_NO_RFC7540_PRIORITIES with + a value of 0 or if the settings parameter was absent, it SHOULD stop + sending PRIORITY_UPDATE frames (Section 7.1), since those frames are + likely to be ignored. However, the client MAY continue sending the + Priority header field (Section 5), as it is an end-to-end signal that + might be useful to nodes behind the server that the client is + directly connected to. + +3. Applicability of the Extensible Priority Scheme + + The priority scheme defined by this document is primarily focused on + the prioritization of HTTP response messages (see Section 3.4 of + [HTTP]). It defines new priority parameters (Section 4) and a means + of conveying those parameters (Sections 5 and 7), which is intended + to communicate the priority of responses to a server that is + responsible for prioritizing them. Section 10 provides + considerations for servers about acting on those signals in + combination with other inputs and factors. + + The CONNECT method (see Section 9.3.6 of [HTTP]) can be used to + establish tunnels. Signaling applies similarly to tunnels; + additional considerations for server prioritization are given in + Section 11. + + Section 9 describes how clients can optionally apply elements of this + scheme locally to the request messages that they generate. + + Some forms of HTTP extensions might change HTTP/2 or HTTP/3 stream + behavior or define new data carriage mechanisms. Such extensions can + themselves define how this priority scheme is to be applied. + +4. Priority Parameters + + The priority information is a sequence of key-value pairs, providing + room for future extensions. Each key-value pair represents a + priority parameter. + + The Priority HTTP header field (Section 5) is an end-to-end way to + transmit this set of priority parameters when a request or a response + is issued. After sending a request, a client can change their view + of response priority (Section 6) by sending HTTP-version-specific + PRIORITY_UPDATE frames as defined in Sections 7.1 and 7.2. Frames + transmit priority parameters on a single hop only. + + Intermediaries can consume and produce priority signals in a + PRIORITY_UPDATE frame or Priority header field. An intermediary that + passes only the Priority request header field to the next hop + preserves the original end-to-end signal from the client; see + Section 14. An intermediary could pass the Priority header field and + additionally send a PRIORITY_UPDATE frame. This would have the + effect of preserving the original client end-to-end signal, while + instructing the next hop to use a different priority, per the + guidance in Section 7. An intermediary that replaces or adds a + Priority request header field overrides the original client end-to- + end signal, which can affect prioritization for all subsequent + recipients of the request. + + For both the Priority header field and the PRIORITY_UPDATE frame, the + set of priority parameters is encoded as a Dictionary (see + Section 3.2 of [STRUCTURED-FIELDS]). + + This document defines the urgency (u) and incremental (i) priority + parameters. When receiving an HTTP request that does not carry these + priority parameters, a server SHOULD act as if their default values + were specified. + + An intermediary can combine signals from requests and responses that + it forwards. Note that omission of priority parameters in responses + is handled differently from omission in requests; see Section 8. + + Receivers parse the Dictionary as described in Section 4.2 of + [STRUCTURED-FIELDS]. Where the Dictionary is successfully parsed, + this document places the additional requirement that unknown priority + parameters, priority parameters with out-of-range values, or values + of unexpected types MUST be ignored. + +4.1. Urgency + + The urgency (u) parameter value is Integer (see Section 3.3.1 of + [STRUCTURED-FIELDS]), between 0 and 7 inclusive, in descending order + of priority. The default is 3. + + Endpoints use this parameter to communicate their view of the + precedence of HTTP responses. The chosen value of urgency can be + based on the expectation that servers might use this information to + transmit HTTP responses in the order of their urgency. The smaller + the value, the higher the precedence. + + The following example shows a request for a CSS file with the urgency + set to 0: + + :method = GET + :scheme = https + :authority = example.net + :path = /style.css + priority = u=0 + + A client that fetches a document that likely consists of multiple + HTTP resources (e.g., HTML) SHOULD assign the default urgency level + to the main resource. This convention allows servers to refine the + urgency using knowledge specific to the website (see Section 8). + + The lowest urgency level (7) is reserved for background tasks such as + delivery of software updates. This urgency level SHOULD NOT be used + for fetching responses that have any impact on user interaction. + +4.2. Incremental + + The incremental (i) parameter value is Boolean (see Section 3.3.6 of + [STRUCTURED-FIELDS]). It indicates if an HTTP response can be + processed incrementally, i.e., provide some meaningful output as + chunks of the response arrive. + + The default value of the incremental parameter is false (0). + + If a client makes concurrent requests with the incremental parameter + set to false, there is no benefit in serving responses with the same + urgency concurrently because the client is not going to process those + responses incrementally. Serving non-incremental responses with the + same urgency one by one, in the order in which those requests were + generated, is considered to be the best strategy. + + If a client makes concurrent requests with the incremental parameter + set to true, serving requests with the same urgency concurrently + might be beneficial. Doing this distributes the connection + bandwidth, meaning that responses take longer to complete. + Incremental delivery is most useful where multiple partial responses + might provide some value to clients ahead of a complete response + being available. + + The following example shows a request for a JPEG file with the + urgency parameter set to 5 and the incremental parameter set to true. + + :method = GET + :scheme = https + :authority = example.net + :path = /image.jpg + priority = u=5, i + +4.3. Defining New Priority Parameters + + When attempting to define new priority parameters, care must be taken + so that they do not adversely interfere with prioritization performed + by existing endpoints or intermediaries that do not understand the + newly defined priority parameters. Since unknown priority parameters + are ignored, new priority parameters should not change the + interpretation of, or modify, the urgency (see Section 4.1) or + incremental (see Section 4.2) priority parameters in a way that is + not backwards compatible or fallback safe. + + For example, if there is a need to provide more granularity than + eight urgency levels, it would be possible to subdivide the range + using an additional priority parameter. Implementations that do not + recognize the parameter can safely continue to use the less granular + eight levels. + + Alternatively, the urgency can be augmented. For example, a + graphical user agent could send a visible priority parameter to + indicate if the resource being requested is within the viewport. + + Generic priority parameters are preferred over vendor-specific, + application-specific, or deployment-specific values. If a generic + value cannot be agreed upon in the community, the parameter's name + should be correspondingly specific (e.g., with a prefix that + identifies the vendor, application, or deployment). + +4.3.1. Registration + + New priority parameters can be defined by registering them in the + "HTTP Priority" registry. This registry governs the keys (short + textual strings) used in the Dictionary (see Section 3.2 of + [STRUCTURED-FIELDS]). Since each HTTP request can have associated + priority signals, there is value in having short key lengths, + especially single-character strings. In order to encourage + extensions while avoiding unintended conflict among attractive key + values, the "HTTP Priority" registry operates two registration + policies, depending on key length. + + * Registration requests for priority parameters with a key length of + one use the Specification Required policy, per Section 4.6 of + [RFC8126]. + + * Registration requests for priority parameters with a key length + greater than one use the Expert Review policy, per Section 4.5 of + [RFC8126]. A specification document is appreciated but not + required. + + When reviewing registration requests, the designated expert(s) can + consider the additional guidance provided in Section 4.3 but cannot + use it as a basis for rejection. + + Registration requests should use the following template: + + Name: [a name for the priority parameter that matches the parameter + key] + + Description: [a description of the priority parameter semantics and + value] + + Reference: [to a specification defining this priority parameter] + + See the registry at <https://www.iana.org/assignments/http-priority> + for details on where to send registration requests. + +5. The Priority HTTP Header Field + + The Priority HTTP header field is a Dictionary that carries priority + parameters (see Section 4). It can appear in requests and responses. + It is an end-to-end signal that indicates the endpoint's view of how + HTTP responses should be prioritized. Section 8 describes how + intermediaries can combine the priority information sent from clients + and servers. Clients cannot interpret the appearance or omission of + a Priority response header field as acknowledgement that any + prioritization has occurred. Guidance for how endpoints can act on + Priority header values is given in Sections 9 and 10. + + An HTTP request with a Priority header field might be cached and + reused for subsequent requests; see [CACHING]. When an origin server + generates the Priority response header field based on properties of + an HTTP request it receives, the server is expected to control the + cacheability or the applicability of the cached response by using + header fields that control the caching behavior (e.g., Cache-Control, + Vary). + +6. Reprioritization + + After a client sends a request, it may be beneficial to change the + priority of the response. As an example, a web browser might issue a + prefetch request for a JavaScript file with the urgency parameter of + the Priority request header field set to u=7 (background). Then, + when the user navigates to a page that references the new JavaScript + file, while the prefetch is in progress, the browser would send a + reprioritization signal with the Priority Field Value set to u=0. + The PRIORITY_UPDATE frame (Section 7) can be used for such + reprioritization. + +7. The PRIORITY_UPDATE Frame + + This document specifies a new PRIORITY_UPDATE frame for HTTP/2 + [HTTP/2] and HTTP/3 [HTTP/3]. It carries priority parameters and + references the target of the prioritization based on a version- + specific identifier. In HTTP/2, this identifier is the stream ID; in + HTTP/3, the identifier is either the stream ID or push ID. Unlike + the Priority header field, the PRIORITY_UPDATE frame is a hop-by-hop + signal. + + PRIORITY_UPDATE frames are sent by clients on the control stream, + allowing them to be sent independently of the stream that carries the + response. This means they can be used to reprioritize a response or + a push stream, or to signal the initial priority of a response + instead of the Priority header field. + + A PRIORITY_UPDATE frame communicates a complete set of all priority + parameters in the Priority Field Value field. Omitting a priority + parameter is a signal to use its default value. Failure to parse the + Priority Field Value MAY be treated as a connection error. In + HTTP/2, the error is of type PROTOCOL_ERROR; in HTTP/3, the error is + of type H3_GENERAL_PROTOCOL_ERROR. + + A client MAY send a PRIORITY_UPDATE frame before the stream that it + references is open (except for HTTP/2 push streams; see Section 7.1). + Furthermore, HTTP/3 offers no guaranteed ordering across streams, + which could cause the frame to be received earlier than intended. + Either case leads to a race condition where a server receives a + PRIORITY_UPDATE frame that references a request stream that is yet to + be opened. To solve this condition, for the purposes of scheduling, + the most recently received PRIORITY_UPDATE frame can be considered as + the most up-to-date information that overrides any other signal. + Servers SHOULD buffer the most recently received PRIORITY_UPDATE + frame and apply it once the referenced stream is opened. Holding + PRIORITY_UPDATE frames for each stream requires server resources, + which can be bounded by local implementation policy. Although there + is no limit to the number of PRIORITY_UPDATE frames that can be sent, + storing only the most recently received frame limits resource + commitment. + +7.1. HTTP/2 PRIORITY_UPDATE Frame + + The HTTP/2 PRIORITY_UPDATE frame (type=0x10) is used by clients to + signal the initial priority of a response, or to reprioritize a + response or push stream. It carries the stream ID of the response + and the priority in ASCII text, using the same representation as the + Priority header field value. + + The Stream Identifier field (see Section 5.1.1 of [HTTP/2]) in the + PRIORITY_UPDATE frame header MUST be zero (0x0). Receiving a + PRIORITY_UPDATE frame with a field of any other value MUST be treated + as a connection error of type PROTOCOL_ERROR. + + HTTP/2 PRIORITY_UPDATE Frame { + Length (24), + Type (8) = 0x10, + + Unused Flags (8), + + Reserved (1), + Stream Identifier (31), + + Reserved (1), + Prioritized Stream ID (31), + Priority Field Value (..), + } + + Figure 1: HTTP/2 PRIORITY_UPDATE Frame Format + + The Length, Type, Unused Flag(s), Reserved, and Stream Identifier + fields are described in Section 4 of [HTTP/2]. The PRIORITY_UPDATE + frame payload contains the following additional fields: + + Prioritized Stream ID: A 31-bit stream identifier for the stream + that is the target of the priority update. + + Priority Field Value: The priority update value in ASCII text, + encoded using Structured Fields. This is the same representation + as the Priority header field value. + + When the PRIORITY_UPDATE frame applies to a request stream, clients + SHOULD provide a prioritized stream ID that refers to a stream in the + "open", "half-closed (local)", or "idle" state (i.e., streams where + data might still be received). Servers can discard frames where the + prioritized stream ID refers to a stream in the "half-closed (local)" + or "closed" state (i.e., streams where no further data will be sent). + The number of streams that have been prioritized but remain in the + "idle" state plus the number of active streams (those in the "open" + state or in either of the "half-closed" states; see Section 5.1.2 of + [HTTP/2]) MUST NOT exceed the value of the + SETTINGS_MAX_CONCURRENT_STREAMS parameter. Servers that receive such + a PRIORITY_UPDATE MUST respond with a connection error of type + PROTOCOL_ERROR. + + When the PRIORITY_UPDATE frame applies to a push stream, clients + SHOULD provide a prioritized stream ID that refers to a stream in the + "reserved (remote)" or "half-closed (local)" state. Servers can + discard frames where the prioritized stream ID refers to a stream in + the "closed" state. Clients MUST NOT provide a prioritized stream ID + that refers to a push stream in the "idle" state. Servers that + receive a PRIORITY_UPDATE for a push stream in the "idle" state MUST + respond with a connection error of type PROTOCOL_ERROR. + + If a PRIORITY_UPDATE frame is received with a prioritized stream ID + of 0x0, the recipient MUST respond with a connection error of type + PROTOCOL_ERROR. + + Servers MUST NOT send PRIORITY_UPDATE frames. If a client receives a + PRIORITY_UPDATE frame, it MUST respond with a connection error of + type PROTOCOL_ERROR. + +7.2. HTTP/3 PRIORITY_UPDATE Frame + + The HTTP/3 PRIORITY_UPDATE frame (type=0xF0700 or 0xF0701) is used by + clients to signal the initial priority of a response, or to + reprioritize a response or push stream. It carries the identifier of + the element that is being prioritized and the updated priority in + ASCII text that uses the same representation as that of the Priority + header field value. PRIORITY_UPDATE with a frame type of 0xF0700 is + used for request streams, while PRIORITY_UPDATE with a frame type of + 0xF0701 is used for push streams. + + The PRIORITY_UPDATE frame MUST be sent on the client control stream + (see Section 6.2.1 of [HTTP/3]). Receiving a PRIORITY_UPDATE frame + on a stream other than the client control stream MUST be treated as a + connection error of type H3_FRAME_UNEXPECTED. + + HTTP/3 PRIORITY_UPDATE Frame { + Type (i) = 0xF0700..0xF0701, + Length (i), + Prioritized Element ID (i), + Priority Field Value (..), + } + + Figure 2: HTTP/3 PRIORITY_UPDATE Frame + + The PRIORITY_UPDATE frame payload has the following fields: + + Prioritized Element ID: The stream ID or push ID that is the target + of the priority update. + + Priority Field Value: The priority update value in ASCII text, + encoded using Structured Fields. This is the same representation + as the Priority header field value. + + The request-stream variant of PRIORITY_UPDATE (type=0xF0700) MUST + reference a request stream. If a server receives a PRIORITY_UPDATE + (type=0xF0700) for a stream ID that is not a request stream, this + MUST be treated as a connection error of type H3_ID_ERROR. The + stream ID MUST be within the client-initiated bidirectional stream + limit. If a server receives a PRIORITY_UPDATE (type=0xF0700) with a + stream ID that is beyond the stream limits, this SHOULD be treated as + a connection error of type H3_ID_ERROR. Generating an error is not + mandatory because HTTP/3 implementations might have practical + barriers to determining the active stream concurrency limit that is + applied by the QUIC layer. + + The push-stream variant of PRIORITY_UPDATE (type=0xF0701) MUST + reference a promised push stream. If a server receives a + PRIORITY_UPDATE (type=0xF0701) with a push ID that is greater than + the maximum push ID or that has not yet been promised, this MUST be + treated as a connection error of type H3_ID_ERROR. + + Servers MUST NOT send PRIORITY_UPDATE frames of either type. If a + client receives a PRIORITY_UPDATE frame, this MUST be treated as a + connection error of type H3_FRAME_UNEXPECTED. + +8. Merging Client- and Server-Driven Priority Parameters + + It is not always the case that the client has the best understanding + of how the HTTP responses deserve to be prioritized. The server + might have additional information that can be combined with the + client's indicated priority in order to improve the prioritization of + the response. For example, use of an HTML document might depend + heavily on one of the inline images; the existence of such + dependencies is typically best known to the server. Or, a server + that receives requests for a font [RFC8081] and images with the same + urgency might give higher precedence to the font, so that a visual + client can render textual information at an early moment. + + An origin can use the Priority response header field to indicate its + view on how an HTTP response should be prioritized. An intermediary + that forwards an HTTP response can use the priority parameters found + in the Priority response header field, in combination with the client + Priority request header field, as input to its prioritization + process. No guidance is provided for merging priorities; this is + left as an implementation decision. + + The absence of a priority parameter in an HTTP response indicates the + server's disinterest in changing the client-provided value. This is + different from the request header field, in which omission of a + priority parameter implies the use of its default value (see + Section 4). + + As a non-normative example, when the client sends an HTTP request + with the urgency parameter set to 5 and the incremental parameter set + to true + + :method = GET + :scheme = https + :authority = example.net + :path = /menu.png + priority = u=5, i + + and the origin responds with + + :status = 200 + content-type = image/png + priority = u=1 + + the intermediary might alter its understanding of the urgency from 5 + to 1, because it prefers the server-provided value over the client's. + The incremental value continues to be true, i.e., the value specified + by the client, as the server did not specify the incremental (i) + parameter. + +9. Client Scheduling + + A client MAY use priority values to make local processing or + scheduling choices about the requests it initiates. + +10. Server Scheduling + + It is generally beneficial for an HTTP server to send all responses + as early as possible. However, when serving multiple requests on a + single connection, there could be competition between the requests + for resources such as connection bandwidth. This section describes + considerations regarding how servers can schedule the order in which + the competing responses will be sent when such competition exists. + + Server scheduling is a prioritization process based on many inputs, + with priority signals being only one form of input. Factors such as + implementation choices or deployment environment also play a role. + Any given connection is likely to have many dynamic permutations. + For these reasons, it is not possible to describe a universal + scheduling algorithm. This document provides some basic, non- + exhaustive recommendations for how servers might act on priority + parameters. It does not describe in detail how servers might combine + priority signals with other factors. Endpoints cannot depend on + particular treatment based on priority signals. Expressing priority + is only a suggestion. + + It is RECOMMENDED that, when possible, servers respect the urgency + parameter (Section 4.1), sending higher-urgency responses before + lower-urgency responses. + + The incremental parameter indicates how a client processes response + bytes as they arrive. It is RECOMMENDED that, when possible, servers + respect the incremental parameter (Section 4.2). + + Non-incremental responses of the same urgency SHOULD be served by + prioritizing bandwidth allocation in ascending order of the stream + ID, which corresponds to the order in which clients make requests. + Doing so ensures that clients can use request ordering to influence + response order. + + Incremental responses of the same urgency SHOULD be served by sharing + bandwidth among them. The message content of incremental responses + is used as parts, or chunks, are received. A client might benefit + more from receiving a portion of all these resources rather than the + entirety of a single resource. How large a portion of the resource + is needed to be useful in improving performance varies. Some + resource types place critical elements early; others can use + information progressively. This scheme provides no explicit mandate + about how a server should use size, type, or any other input to + decide how to prioritize. + + There can be scenarios where a server will need to schedule multiple + incremental and non-incremental responses at the same urgency level. + Strictly abiding by the scheduling guidance based on urgency and + request generation order might lead to suboptimal results at the + client, as early non-incremental responses might prevent the serving + of incremental responses issued later. The following are examples of + such challenges: + + 1. At the same urgency level, a non-incremental request for a large + resource followed by an incremental request for a small resource. + + 2. At the same urgency level, an incremental request of + indeterminate length followed by a non-incremental large + resource. + + It is RECOMMENDED that servers avoid such starvation where possible. + The method for doing so is an implementation decision. For example, + a server might preemptively send responses of a particular + incremental type based on other information such as content size. + + Optimal scheduling of server push is difficult, especially when + pushed resources contend with active concurrent requests. Servers + can consider many factors when scheduling, such as the type or size + of resource being pushed, the priority of the request that triggered + the push, the count of active concurrent responses, the priority of + other active concurrent responses, etc. There is no general guidance + on the best way to apply these. A server that is too simple could + easily push at too high a priority and block client requests, or push + at too low a priority and delay the response, negating intended goals + of server push. + + Priority signals are a factor for server push scheduling. The + concept of parameter value defaults applies slightly differently + because there is no explicit client-signaled initial priority. A + server can apply priority signals provided in an origin response; see + the merging guidance given in Section 8. In the absence of origin + signals, applying default parameter values could be suboptimal. By + whatever means a server decides to schedule a pushed response, it can + signal the intended priority to the client by including the Priority + field in a PUSH_PROMISE or HEADERS frame. + +10.1. Intermediaries with Multiple Backend Connections + + An intermediary serving an HTTP connection might split requests over + multiple backend connections. When it applies prioritization rules + strictly, low-priority requests cannot make progress while requests + with higher priorities are in flight. This blocking can propagate to + backend connections, which the peer might interpret as a connection + stall. Endpoints often implement protections against stalls, such as + abruptly closing connections after a certain time period. To reduce + the possibility of this occurring, intermediaries can avoid strictly + following prioritization and instead allocate small amounts of + bandwidth for all the requests that they are forwarding, so that + every request can make some progress over time. + + Similarly, servers SHOULD allocate some amount of bandwidths to + streams acting as tunnels. + +11. Scheduling and the CONNECT Method + + When a stream carries a CONNECT request, the scheduling guidance in + this document applies to the frames on the stream. A client that + issues multiple CONNECT requests can set the incremental parameter to + true. Servers that implement the recommendations for handling of the + incremental parameter (Section 10) are likely to schedule these + fairly, preventing one CONNECT stream from blocking others. + +12. Retransmission Scheduling + + Transport protocols such as TCP and QUIC provide reliability by + detecting packet losses and retransmitting lost information. In + addition to the considerations in Section 10, scheduling of + retransmission data could compete with new data. The remainder of + this section discusses considerations when using QUIC. + + Section 13.3 of [QUIC] states the following: "Endpoints SHOULD + prioritize retransmission of data over sending new data, unless + priorities specified by the application indicate otherwise". When an + HTTP/3 application uses the priority scheme defined in this document + and the QUIC transport implementation supports application-indicated + stream priority, a transport that considers the relative priority of + streams when scheduling both new data and retransmission data might + better match the expectations of the application. However, there are + no requirements on how a transport chooses to schedule based on this + information because the decision depends on several factors and + trade-offs. It could prioritize new data for a higher-urgency stream + over retransmission data for a lower-priority stream, or it could + prioritize retransmission data over new data irrespective of + urgencies. + + Section 6.2.4 of [QUIC-RECOVERY] also highlights considerations + regarding application priorities when sending probe packets after + Probe Timeout timer expiration. A QUIC implementation supporting + application-indicated priorities might use the relative priority of + streams when choosing probe data. + +13. Fairness + + Typically, HTTP implementations depend on the underlying transport to + maintain fairness between connections competing for bandwidth. When + an intermediary receives HTTP requests on client connections, it + forwards them to backend connections. Depending on how the + intermediary coalesces or splits requests across different backend + connections, different clients might experience dissimilar + performance. This dissimilarity might expand if the intermediary + also uses priority signals when forwarding requests. Sections 13.1 + and 13.2 discuss mitigations of this expansion of unfairness. + + Conversely, Section 13.3 discusses how servers might intentionally + allocate unequal bandwidth to some connections, depending on the + priority signals. + +13.1. Coalescing Intermediaries + + When an intermediary coalesces HTTP requests coming from multiple + clients into one HTTP/2 or HTTP/3 connection going to the backend + server, requests that originate from one client might carry signals + indicating higher priority than those coming from others. + + It is sometimes beneficial for the server running behind an + intermediary to obey Priority header field values. As an example, a + resource-constrained server might defer the transmission of software + update files that have the background urgency level (7). However, in + the worst case, the asymmetry between the priority declared by + multiple clients might cause all responses going to one user agent to + be delayed until all responses going to another user agent have been + sent. + + In order to mitigate this fairness problem, a server could use + knowledge about the intermediary as another input in its + prioritization decisions. For instance, if a server knows the + intermediary is coalescing requests, then it could avoid serving the + responses in their entirety and instead distribute bandwidth (for + example, in a round-robin manner). This can work if the constrained + resource is network capacity between the intermediary and the user + agent, as the intermediary buffers responses and forwards the chunks + based on the prioritization scheme it implements. + + A server can determine if a request came from an intermediary through + configuration or can check to see if the request contains one of the + following header fields: + + * Forwarded [FORWARDED], X-Forwarded-For + + * Via (see Section 7.6.3 of [HTTP]) + +13.2. HTTP/1.x Back Ends + + It is common for Content Delivery Network (CDN) infrastructure to + support different HTTP versions on the front end and back end. For + instance, the client-facing edge might support HTTP/2 and HTTP/3 + while communication to backend servers is done using HTTP/1.1. + Unlike connection coalescing, the CDN will "demux" requests into + discrete connections to the back end. Response multiplexing in a + single connection is not supported by HTTP/1.1 (or older), so there + is not a fairness problem. However, backend servers MAY still use + client headers for request scheduling. Backend servers SHOULD only + schedule based on client priority information where that information + can be scoped to individual end clients. Authentication and other + session information might provide this linkability. + +13.3. Intentional Introduction of Unfairness + + It is sometimes beneficial to deprioritize the transmission of one + connection over others, knowing that doing so introduces a certain + amount of unfairness between the connections and therefore between + the requests served on those connections. + + For example, a server might use a scavenging congestion controller on + connections that only convey background priority responses such as + software update images. Doing so improves responsiveness of other + connections at the cost of delaying the delivery of updates. + +14. Why Use an End-to-End Header Field? + + In contrast to the prioritization scheme of HTTP/2, which uses a hop- + by-hop frame, the Priority header field is defined as "end-to-end". + + The way that a client processes a response is a property associated + with the client generating that request, not that of an intermediary. + Therefore, it is an end-to-end property. How these end-to-end + properties carried by the Priority header field affect the + prioritization between the responses that share a connection is a + hop-by-hop issue. + + Having the Priority header field defined as end-to-end is important + for caching intermediaries. Such intermediaries can cache the value + of the Priority header field along with the response and utilize the + value of the cached header field when serving the cached response, + only because the header field is defined as end-to-end rather than + hop-by-hop. + +15. Security Considerations + + Section 7 describes considerations for server buffering of + PRIORITY_UPDATE frames. + + Section 10 presents examples where servers that prioritize responses + in a certain way might be starved of the ability to transmit + responses. + + The security considerations from [STRUCTURED-FIELDS] apply to the + processing of priority parameters defined in Section 4. + +16. IANA Considerations + + This specification registers the following entry in the "Hypertext + Transfer Protocol (HTTP) Field Name Registry" defined in [HTTP/2]: + + Field Name: Priority + Status: permanent + Reference: This document + + This specification registers the following entry in the "HTTP/2 + Settings" registry defined in [HTTP/2]: + + Code: 0x9 + Name: SETTINGS_NO_RFC7540_PRIORITIES + Initial Value: 0 + Reference: This document + + This specification registers the following entry in the "HTTP/2 Frame + Type" registry defined in [HTTP/2]: + + Code: 0x10 + Frame Type: PRIORITY_UPDATE + Reference: This document + + This specification registers the following entry in the "HTTP/3 Frame + Types" registry established by [HTTP/3]: + + Value: 0xF0700-0xF0701 + Frame Type: PRIORITY_UPDATE + Status: permanent + Reference: This document + Change Controller: IETF + Contact: ietf-http-wg@w3.org + + IANA has created the "Hypertext Transfer Protocol (HTTP) Priority" + registry at <https://www.iana.org/assignments/http-priority> and has + populated it with the entries in Table 1; see Section 4.3.1 for its + associated procedures. + + +======+==================================+=============+ + | Name | Description | Reference | + +======+==================================+=============+ + | u | The urgency of an HTTP response. | Section 4.1 | + +------+----------------------------------+-------------+ + | i | Whether an HTTP response can be | Section 4.2 | + | | processed incrementally. | | + +------+----------------------------------+-------------+ + + Table 1: Initial Priority Parameters + +17. References + +17.1. Normative References + + [HTTP] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP Semantics", STD 97, RFC 9110, + DOI 10.17487/RFC9110, June 2022, + <https://www.rfc-editor.org/info/rfc9110>. + + [HTTP/2] Thomson, M., Ed. and C. Benfield, Ed., "HTTP/2", RFC 9113, + DOI 10.17487/RFC9113, June 2022, + <https://www.rfc-editor.org/info/rfc9113>. + + [HTTP/3] Bishop, M., Ed., "HTTP/3", RFC 9114, DOI 10.17487/RFC9114, + June 2022, <https://www.rfc-editor.org/info/rfc9114>. + + [QUIC] Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC8126] Cotton, M., Leiba, B., and T. Narten, "Guidelines for + Writing an IANA Considerations Section in RFCs", BCP 26, + RFC 8126, DOI 10.17487/RFC8126, June 2017, + <https://www.rfc-editor.org/info/rfc8126>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + + [STRUCTURED-FIELDS] + Nottingham, M. and P-H. Kamp, "Structured Field Values for + HTTP", RFC 8941, DOI 10.17487/RFC8941, February 2021, + <https://www.rfc-editor.org/info/rfc8941>. + +17.2. Informative References + + [CACHING] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP Caching", STD 98, RFC 9111, + DOI 10.17487/RFC9111, June 2022, + <https://www.rfc-editor.org/info/rfc9111>. + + [FORWARDED] + Petersson, A. and M. Nilsson, "Forwarded HTTP Extension", + RFC 7239, DOI 10.17487/RFC7239, June 2014, + <https://www.rfc-editor.org/info/rfc7239>. + + [MARX] Marx, R., De Decker, T., Quax, P., and W. Lamotte, "Of the + Utmost Importance: Resource Prioritization in HTTP/3 over + QUIC", SCITEPRESS Proceedings of the 15th International + Conference on Web Information Systems and Technologies + (pages 130-143), DOI 10.5220/0008191701300143, September + 2019, <https://www.doi.org/10.5220/0008191701300143>. + + [PRIORITY-SETTING] + Lassey, B. and L. Pardue, "Declaring Support for HTTP/2 + Priorities", Work in Progress, Internet-Draft, draft- + lassey-priority-setting-00, 25 July 2019, + <https://datatracker.ietf.org/doc/html/draft-lassey- + priority-setting-00>. + + [QUIC-RECOVERY] + Iyengar, J., Ed. and I. Swett, Ed., "QUIC Loss Detection + and Congestion Control", RFC 9002, DOI 10.17487/RFC9002, + May 2021, <https://www.rfc-editor.org/info/rfc9002>. + + [RFC7540] Belshe, M., Peon, R., and M. Thomson, Ed., "Hypertext + Transfer Protocol Version 2 (HTTP/2)", RFC 7540, + DOI 10.17487/RFC7540, May 2015, + <https://www.rfc-editor.org/info/rfc7540>. + + [RFC8081] Lilley, C., "The "font" Top-Level Media Type", RFC 8081, + DOI 10.17487/RFC8081, February 2017, + <https://www.rfc-editor.org/info/rfc8081>. + +Acknowledgements + + Roy Fielding presented the idea of using a header field for + representing priorities in + <https://www.ietf.org/proceedings/83/slides/slides-83-httpbis-5.pdf>. + In <https://github.com/pmeenan/http3-prioritization-proposal>, + Patrick Meenan advocated for representing the priorities using a + tuple of urgency and concurrency. The ability to disable HTTP/2 + prioritization is inspired by [PRIORITY-SETTING], authored by Brad + Lassey and Lucas Pardue, with modifications based on feedback that + was not incorporated into an update to that document. + + The motivation for defining an alternative to HTTP/2 priorities is + drawn from discussion within the broad HTTP community. Special + thanks to Roberto Peon, Martin Thomson, and Netflix for text that was + incorporated explicitly in this document. + + In addition to the people above, this document owes a lot to the + extensive discussion in the HTTP priority design team, consisting of + Alan Frindell, Andrew Galloni, Craig Taylor, Ian Swett, Matthew Cox, + Mike Bishop, Roberto Peon, Robin Marx, Roy Fielding, and the authors + of this document. + + Yang Chi contributed the section on retransmission scheduling. + +Authors' Addresses + + Kazuho Oku + Fastly + Email: kazuhooku@gmail.com + + Additional contact information: + + ๅฅฅ ไธ€็ฉ‚ + Fastly + + + Lucas Pardue + Cloudflare + Email: lucaspardue.24.7@gmail.com diff --git a/eval/corpora/rfc/RFC 9220 - Bootstrapping WebSockets with HTTP3.txt b/eval/corpora/rfc/RFC 9220 - Bootstrapping WebSockets with HTTP3.txt new file mode 100644 index 00000000..eb86c693 --- /dev/null +++ b/eval/corpora/rfc/RFC 9220 - Bootstrapping WebSockets with HTTP3.txt @@ -0,0 +1,168 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) R. Hamilton +Request for Comments: 9220 Google +Category: Standards Track June 2022 +ISSN: 2070-1721 + + + Bootstrapping WebSockets with HTTP/3 + +Abstract + + The mechanism for running the WebSocket Protocol over a single stream + of an HTTP/2 connection is equally applicable to HTTP/3, but the + HTTP-version-specific details need to be specified. This document + describes how the mechanism is adapted for HTTP/3. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9220. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 2. Conventions and Definitions + 3. WebSockets Upgrade over HTTP/3 + 4. Security Considerations + 5. IANA Considerations + 6. Normative References + Acknowledgments + Author's Address + +1. Introduction + + "Bootstrapping WebSockets with HTTP/2" [RFC8441] defines an extension + to HTTP/2 [HTTP/2] that is also useful in HTTP/3 [HTTP/3]. This + extension makes use of an HTTP/2 setting. Appendix A.3 of [HTTP/3] + gives some guidance on what changes (if any) are appropriate when + porting settings from HTTP/2 to HTTP/3. + +2. Conventions and Definitions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + +3. WebSockets Upgrade over HTTP/3 + + [RFC8441] defines a mechanism for running the WebSocket Protocol + [RFC6455] over a single stream of an HTTP/2 connection. It defines + an Extended CONNECT method that specifies a new ":protocol" pseudo- + header field and new semantics for the ":path" and ":authority" + pseudo-header fields. It also defines a new HTTP/2 setting sent by a + server to allow the client to use Extended CONNECT. + + The semantics of the pseudo-header fields and setting are identical + to those in HTTP/2 as defined in [RFC8441]. Appendix A.3 of [HTTP/3] + requires that HTTP/3 settings be registered separately for HTTP/3. + The SETTINGS_ENABLE_CONNECT_PROTOCOL value is 0x08 (decimal 8), as in + HTTP/2. + + If a server advertises support for Extended CONNECT but receives an + Extended CONNECT request with a ":protocol" value that is unknown or + is not supported, the server SHOULD respond to the request with a 501 + (Not Implemented) status code (Section 15.6.2 of [HTTP]). A server + MAY provide more information via a "problem details" response + [RFC7807]. + + The HTTP/3 stream closure is also analogous to the TCP connection + closure of [RFC6455]. Orderly TCP-level closures are represented as + a FIN bit on the stream (Section 4.4 of [HTTP/3]). RST exceptions + are represented with a stream error (Section 8 of [HTTP/3]) of type + H3_REQUEST_CANCELLED (Section 8.1 of [HTTP/3]). + +4. Security Considerations + + This document introduces no new security considerations beyond those + discussed in [RFC8441]. + +5. IANA Considerations + + This document registers a new setting in the "HTTP/3 Settings" + registry (Section 11.2.2 of [HTTP/3]). + + Value: 0x08 + Setting Name: SETTINGS_ENABLE_CONNECT_PROTOCOL + Default: 0 + Status: permanent + Specification: This document + Change Controller: IETF + Contact: HTTP Working Group (ietf-http-wg@w3.org) + +6. Normative References + + [HTTP] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP Semantics", STD 97, RFC 9110, + DOI 10.17487/RFC9110, June 2022, + <https://www.rfc-editor.org/info/rfc9110>. + + [HTTP/2] Thomson, M., Ed. and C. Benfield, Ed., "HTTP/2", RFC 9113, + DOI 10.17487/RFC9113, June 2022, + <https://www.rfc-editor.org/info/rfc9113>. + + [HTTP/3] Bishop, M., Ed., "HTTP/3", RFC 9114, DOI 10.17487/RFC9114, + June 2022, <https://www.rfc-editor.org/info/rfc9114>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC6455] Fette, I. and A. Melnikov, "The WebSocket Protocol", + RFC 6455, DOI 10.17487/RFC6455, December 2011, + <https://www.rfc-editor.org/info/rfc6455>. + + [RFC7807] Nottingham, M. and E. Wilde, "Problem Details for HTTP + APIs", RFC 7807, DOI 10.17487/RFC7807, March 2016, + <https://www.rfc-editor.org/info/rfc7807>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + + [RFC8441] McManus, P., "Bootstrapping WebSockets with HTTP/2", + RFC 8441, DOI 10.17487/RFC8441, September 2018, + <https://www.rfc-editor.org/info/rfc8441>. + +Acknowledgments + + This document had reviews and input from many contributors in the + IETF HTTP and QUIC Working Groups, with substantive input from David + Schinazi, Martin Thomson, Lucas Pardue, Mike Bishop, Dragana + Damjanovic, Mark Nottingham, and Julian Reschke. + +Author's Address + + Ryan Hamilton + Google + Email: rch@google.com diff --git a/eval/corpora/rfc/RFC 9221 - An Unreliable Datagram Extension to QUIC.txt b/eval/corpora/rfc/RFC 9221 - An Unreliable Datagram Extension to QUIC.txt new file mode 100644 index 00000000..18ca4fa5 --- /dev/null +++ b/eval/corpora/rfc/RFC 9221 - An Unreliable Datagram Extension to QUIC.txt @@ -0,0 +1,435 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) T. Pauly +Request for Comments: 9221 E. Kinnear +Category: Standards Track Apple Inc. +ISSN: 2070-1721 D. Schinazi + Google LLC + March 2022 + + + An Unreliable Datagram Extension to QUIC + +Abstract + + This document defines an extension to the QUIC transport protocol to + add support for sending and receiving unreliable datagrams over a + QUIC connection. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9221. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 1.1. Specification of Requirements + 2. Motivation + 3. Transport Parameter + 4. Datagram Frame Types + 5. Behavior and Usage + 5.1. Multiplexing Datagrams + 5.2. Acknowledgement Handling + 5.3. Flow Control + 5.4. Congestion Control + 6. Security Considerations + 7. IANA Considerations + 7.1. QUIC Transport Parameter + 7.2. QUIC Frame Types + 8. References + 8.1. Normative References + 8.2. Informative References + Acknowledgments + Authors' Addresses + +1. Introduction + + The QUIC transport protocol [RFC9000] provides a secure, multiplexed + connection for transmitting reliable streams of application data. + QUIC uses various frame types to transmit data within packets, and + each frame type defines whether the data it contains will be + retransmitted. Streams of reliable application data are sent using + STREAM frames. + + Some applications, particularly those that need to transmit real-time + data, prefer to transmit data unreliably. In the past, these + applications have built directly upon UDP [RFC0768] as a transport + and have often added security with DTLS [RFC6347]. Extending QUIC to + support transmitting unreliable application data provides another + option for secure datagrams with the added benefit of sharing the + cryptographic and authentication context used for reliable streams. + + This document defines two new DATAGRAM QUIC frame types that carry + application data without requiring retransmissions. + +1.1. Specification of Requirements + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + +2. Motivation + + Transmitting unreliable data over QUIC provides benefits over + existing solutions: + + * Applications that want to use both a reliable stream and an + unreliable flow to the same peer can benefit by sharing a single + handshake and authentication context between a reliable QUIC + stream and a flow of unreliable QUIC datagrams. This can reduce + the latency required for handshakes compared to opening both a TLS + connection and a DTLS connection. + + * QUIC uses a more nuanced loss recovery mechanism than the DTLS + handshake. This can allow loss recovery to occur more quickly for + QUIC data. + + * QUIC datagrams are subject to QUIC congestion control. Providing + a single congestion control for both reliable and unreliable data + can be more effective and efficient. + + These features can be useful for optimizing audio/video streaming + applications, gaming applications, and other real-time network + applications. + + Unreliable QUIC datagrams can also be used to implement an IP packet + tunnel over QUIC, such as for a Virtual Private Network (VPN). + Internet-layer tunneling protocols generally require a reliable and + authenticated handshake followed by unreliable secure transmission of + IP packets. This can, for example, require a TLS connection for the + control data and DTLS for tunneling IP packets. A single QUIC + connection could support both parts with the use of unreliable + datagrams in addition to reliable streams. + +3. Transport Parameter + + Support for receiving the DATAGRAM frame types is advertised by means + of a QUIC transport parameter (name=max_datagram_frame_size, + value=0x20). The max_datagram_frame_size transport parameter is an + integer value (represented as a variable-length integer) that + represents the maximum size of a DATAGRAM frame (including the frame + type, length, and payload) the endpoint is willing to receive, in + bytes. + + The default for this parameter is 0, which indicates that the + endpoint does not support DATAGRAM frames. A value greater than 0 + indicates that the endpoint supports the DATAGRAM frame types and is + willing to receive such frames on this connection. + + An endpoint MUST NOT send DATAGRAM frames until it has received the + max_datagram_frame_size transport parameter with a non-zero value + during the handshake (or during a previous handshake if 0-RTT is + used). An endpoint MUST NOT send DATAGRAM frames that are larger + than the max_datagram_frame_size value it has received from its peer. + An endpoint that receives a DATAGRAM frame when it has not indicated + support via the transport parameter MUST terminate the connection + with an error of type PROTOCOL_VIOLATION. Similarly, an endpoint + that receives a DATAGRAM frame that is larger than the value it sent + in its max_datagram_frame_size transport parameter MUST terminate the + connection with an error of type PROTOCOL_VIOLATION. + + For most uses of DATAGRAM frames, it is RECOMMENDED to send a value + of 65535 in the max_datagram_frame_size transport parameter to + indicate that this endpoint will accept any DATAGRAM frame that fits + inside a QUIC packet. + + The max_datagram_frame_size transport parameter is a unidirectional + limit and indication of support of DATAGRAM frames. Application + protocols that use DATAGRAM frames MAY choose to only negotiate and + use them in a single direction. + + When clients use 0-RTT, they MAY store the value of the server's + max_datagram_frame_size transport parameter. Doing so allows the + client to send DATAGRAM frames in 0-RTT packets. When servers decide + to accept 0-RTT data, they MUST send a max_datagram_frame_size + transport parameter greater than or equal to the value they sent to + the client in the connection where they sent them the + NewSessionTicket message. If a client stores the value of the + max_datagram_frame_size transport parameter with their 0-RTT state, + they MUST validate that the new value of the max_datagram_frame_size + transport parameter sent by the server in the handshake is greater + than or equal to the stored value; if not, the client MUST terminate + the connection with error PROTOCOL_VIOLATION. + + Application protocols that use datagrams MUST define how they react + to the absence of the max_datagram_frame_size transport parameter. + If datagram support is integral to the application, the application + protocol can fail the handshake if the max_datagram_frame_size + transport parameter is not present. + +4. Datagram Frame Types + + DATAGRAM frames are used to transmit application data in an + unreliable manner. The Type field in the DATAGRAM frame takes the + form 0b0011000X (or the values 0x30 and 0x31). The least significant + bit of the Type field in the DATAGRAM frame is the LEN bit (0x01), + which indicates whether there is a Length field present: if this bit + is set to 0, the Length field is absent and the Datagram Data field + extends to the end of the packet; if this bit is set to 1, the Length + field is present. + + DATAGRAM frames are structured as follows: + + DATAGRAM Frame { + Type (i) = 0x30..0x31, + [Length (i)], + Datagram Data (..), + } + + Figure 1: DATAGRAM Frame Format + + DATAGRAM frames contain the following fields: + + Length: A variable-length integer specifying the length of the + Datagram Data field in bytes. This field is present only when the + LEN bit is set to 1. When the LEN bit is set to 0, the Datagram + Data field extends to the end of the QUIC packet. Note that empty + (i.e., zero-length) datagrams are allowed. + + Datagram Data: The bytes of the datagram to be delivered. + +5. Behavior and Usage + + When an application sends a datagram over a QUIC connection, QUIC + will generate a new DATAGRAM frame and send it in the first available + packet. This frame SHOULD be sent as soon as possible (as determined + by factors like congestion control, described below) and MAY be + coalesced with other frames. + + When a QUIC endpoint receives a valid DATAGRAM frame, it SHOULD + deliver the data to the application immediately, as long as it is + able to process the frame and can store the contents in memory. + + Like STREAM frames, DATAGRAM frames contain application data and MUST + be protected with either 0-RTT or 1-RTT keys. + + Note that while the max_datagram_frame_size transport parameter + places a limit on the maximum size of DATAGRAM frames, that limit can + be further reduced by the max_udp_payload_size transport parameter + and the Maximum Transmission Unit (MTU) of the path between + endpoints. DATAGRAM frames cannot be fragmented; therefore, + application protocols need to handle cases where the maximum datagram + size is limited by other factors. + +5.1. Multiplexing Datagrams + + DATAGRAM frames belong to a QUIC connection as a whole and are not + associated with any stream ID at the QUIC layer. However, it is + expected that applications will want to differentiate between + specific DATAGRAM frames by using identifiers, such as for logical + flows of datagrams or to distinguish between different kinds of + datagrams. + + Defining the identifiers used to multiplex different kinds of + datagrams or flows of datagrams is the responsibility of the + application protocol running over QUIC. The application defines the + semantics of the Datagram Data field and how it is parsed. + + If the application needs to support the coexistence of multiple flows + of datagrams, one recommended pattern is to use a variable-length + integer at the beginning of the Datagram Data field. This is a + simple approach that allows a large number of flows to be encoded + using minimal space. + + QUIC implementations SHOULD present an API to applications to assign + relative priorities to DATAGRAM frames with respect to each other and + to QUIC streams. + +5.2. Acknowledgement Handling + + Although DATAGRAM frames are not retransmitted upon loss detection, + they are ack-eliciting ([RFC9002]). Receivers SHOULD support + delaying ACK frames (within the limits specified by max_ack_delay) in + response to receiving packets that only contain DATAGRAM frames, + since the sender takes no action if these packets are temporarily + unacknowledged. Receivers will continue to send ACK frames when + conditions indicate a packet might be lost, since the packet's + payload is unknown to the receiver, and when dictated by + max_ack_delay or other protocol components. + + As with any ack-eliciting frame, when a sender suspects that a packet + containing only DATAGRAM frames has been lost, it sends probe packets + to elicit a faster acknowledgement as described in Section 6.2.4 of + [RFC9002]. + + If a sender detects that a packet containing a specific DATAGRAM + frame might have been lost, the implementation MAY notify the + application that it believes the datagram was lost. + + Similarly, if a packet containing a DATAGRAM frame is acknowledged, + the implementation MAY notify the sender application that the + datagram was successfully transmitted and received. Due to + reordering, this can include a DATAGRAM frame that was thought to be + lost but, at a later point, was received and acknowledged. It is + important to note that acknowledgement of a DATAGRAM frame only + indicates that the transport-layer handling on the receiver processed + the frame and does not guarantee that the application on the receiver + successfully processed the data. Thus, this signal cannot replace + application-layer signals that indicate successful processing. + +5.3. Flow Control + + DATAGRAM frames do not provide any explicit flow control signaling + and do not contribute to any per-flow or connection-wide data limit. + + The risk associated with not providing flow control for DATAGRAM + frames is that a receiver might not be able to commit the necessary + resources to process the frames. For example, it might not be able + to store the frame contents in memory. However, since DATAGRAM + frames are inherently unreliable, they MAY be dropped by the receiver + if the receiver cannot process them. + +5.4. Congestion Control + + DATAGRAM frames employ the QUIC connection's congestion controller. + As a result, a connection might be unable to send a DATAGRAM frame + generated by the application until the congestion controller allows + it [RFC9002]. The sender MUST either delay sending the frame until + the controller allows it or drop the frame without sending it (at + which point it MAY notify the application). Implementations that use + packet pacing (Section 7.7 of [RFC9002]) can also delay the sending + of DATAGRAM frames to maintain consistent packet pacing. + + Implementations can optionally support allowing the application to + specify a sending expiration time beyond which a congestion- + controlled DATAGRAM frame ought to be dropped without transmission. + +6. Security Considerations + + The DATAGRAM frame shares the same security properties as the rest of + the data transmitted within a QUIC connection, and the security + considerations of [RFC9000] apply accordingly. All application data + transmitted with the DATAGRAM frame, like the STREAM frame, MUST be + protected either by 0-RTT or 1-RTT keys. + + Application protocols that allow DATAGRAM frames to be sent in 0-RTT + require a profile that defines acceptable use of 0-RTT; see + Section 5.6 of [RFC9001]. + + The use of DATAGRAM frames might be detectable by an adversary on + path that is capable of dropping packets. Since DATAGRAM frames do + not use transport-level retransmission, connections that use DATAGRAM + frames might be distinguished from other connections due to their + different response to packet loss. + +7. IANA Considerations + +7.1. QUIC Transport Parameter + + This document registers a new value in the "QUIC Transport + Parameters" registry maintained at <https://www.iana.org/assignments/ + quic>. + + Value: 0x20 + Parameter Name: max_datagram_frame_size + Status: permanent + Specification: RFC 9221 + +7.2. QUIC Frame Types + + This document registers two new values in the "QUIC Frame Types" + registry maintained at <https://www.iana.org/assignments/quic>. + + Value: 0x30-0x31 + Frame Name: DATAGRAM + Status: permanent + Specification: RFC 9221 + +8. References + +8.1. Normative References + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + + [RFC9000] Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [RFC9001] Thomson, M., Ed. and S. Turner, Ed., "Using TLS to Secure + QUIC", RFC 9001, DOI 10.17487/RFC9001, May 2021, + <https://www.rfc-editor.org/info/rfc9001>. + + [RFC9002] Iyengar, J., Ed. and I. Swett, Ed., "QUIC Loss Detection + and Congestion Control", RFC 9002, DOI 10.17487/RFC9002, + May 2021, <https://www.rfc-editor.org/info/rfc9002>. + +8.2. Informative References + + [RFC0768] Postel, J., "User Datagram Protocol", STD 6, RFC 768, + DOI 10.17487/RFC0768, August 1980, + <https://www.rfc-editor.org/info/rfc768>. + + [RFC6347] Rescorla, E. and N. Modadugu, "Datagram Transport Layer + Security Version 1.2", RFC 6347, DOI 10.17487/RFC6347, + January 2012, <https://www.rfc-editor.org/info/rfc6347>. + +Acknowledgments + + The original proposal for this work came from Ian Swett. + + This document had reviews and input from many contributors in the + IETF QUIC Working Group, with substantive input from Nick Banks, + Lucas Pardue, Rui Paulo, Martin Thomson, Victor Vasiliev, and Chris + Wood. + +Authors' Addresses + + Tommy Pauly + Apple Inc. + One Apple Park Way + Cupertino, CA 95014 + United States of America + Email: tpauly@apple.com + + + Eric Kinnear + Apple Inc. + One Apple Park Way + Cupertino, CA 95014 + United States of America + Email: ekinnear@apple.com + + + David Schinazi + Google LLC + 1600 Amphitheatre Parkway + Mountain View, CA 94043 + United States of America + Email: dschinazi.ietf@gmail.com diff --git a/eval/corpora/rfc/RFC 9250 - DNS over Dedicated QUIC Connections.txt b/eval/corpora/rfc/RFC 9250 - DNS over Dedicated QUIC Connections.txt new file mode 100644 index 00000000..1aed862d --- /dev/null +++ b/eval/corpora/rfc/RFC 9250 - DNS over Dedicated QUIC Connections.txt @@ -0,0 +1,1450 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) C. Huitema +Request for Comments: 9250 Private Octopus Inc. +Category: Standards Track S. Dickinson +ISSN: 2070-1721 Sinodun IT + A. Mankin + Salesforce + May 2022 + + + DNS over Dedicated QUIC Connections + +Abstract + + This document describes the use of QUIC to provide transport + confidentiality for DNS. The encryption provided by QUIC has similar + properties to those provided by TLS, while QUIC transport eliminates + the head-of-line blocking issues inherent with TCP and provides more + efficient packet-loss recovery than UDP. DNS over QUIC (DoQ) has + privacy properties similar to DNS over TLS (DoT) specified in RFC + 7858, and latency characteristics similar to classic DNS over UDP. + This specification describes the use of DoQ as a general-purpose + transport for DNS and includes the use of DoQ for stub to recursive, + recursive to authoritative, and zone transfer scenarios. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9250. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 2. Key Words + 3. Design Considerations + 3.1. Provide DNS Privacy + 3.2. Design for Minimum Latency + 3.3. Middlebox Considerations + 3.4. No Server-Initiated Transactions + 4. Specifications + 4.1. Connection Establishment + 4.1.1. Port Selection + 4.2. Stream Mapping and Usage + 4.2.1. DNS Message IDs + 4.3. DoQ Error Codes + 4.3.1. Transaction Cancellation + 4.3.2. Transaction Errors + 4.3.3. Protocol Errors + 4.3.4. Alternative Error Codes + 4.4. Connection Management + 4.5. Session Resumption and 0-RTT + 4.6. Message Sizes + 5. Implementation Requirements + 5.1. Authentication + 5.2. Fallback to Other Protocols on Connection Failure + 5.3. Address Validation + 5.4. Padding + 5.5. Connection Handling + 5.5.1. Connection Reuse + 5.5.2. Resource Management + 5.5.3. Using 0-RTT and Session Resumption + 5.5.4. Controlling Connection Migration for Privacy + 5.6. Processing Queries in Parallel + 5.7. Zone Transfer + 5.8. Flow Control Mechanisms + 6. Security Considerations + 7. Privacy Considerations + 7.1. Privacy Issues with 0-RTT data + 7.2. Privacy Issues with Session Resumption + 7.3. Privacy Issues with Address Validation Tokens + 7.4. Privacy Issues with Long Duration Sessions + 7.5. Traffic Analysis + 8. IANA Considerations + 8.1. Registration of a DoQ Identification String + 8.2. Reservation of a Dedicated Port + 8.3. Reservation of an Extended DNS Error Code: Too Early + 8.4. DNS-over-QUIC Error Codes Registry + 9. References + 9.1. Normative References + 9.2. Informative References + Appendix A. The NOTIFY Service + Acknowledgements + Authors' Addresses + +1. Introduction + + Domain Name System (DNS) concepts are specified in "Domain names - + concepts and facilities" [RFC1034]. The transmission of DNS queries + and responses over UDP and TCP is specified in "Domain names - + implementation and specification" [RFC1035]. + + This document presents a mapping of the DNS protocol over the QUIC + transport [RFC9000] [RFC9001]. DNS over QUIC is referred to here as + DoQ, in line with "DNS Terminology" [DNS-TERMS]. + + The goals of the DoQ mapping are: + + 1. Provide the same DNS privacy protection as DoT [RFC7858]. This + includes an option for the client to authenticate the server by + means of an authentication domain name as specified in "Usage + Profiles for DNS over TLS and DNS over DTLS" [RFC8310]. + + 2. Provide an improved level of source address validation for DNS + servers compared to classic DNS over UDP. + + 3. Provide a transport that does not impose path MTU limitations on + the size of DNS responses it can send. + + In order to achieve these goals, and to support ongoing work on + encryption of DNS, the scope of this document includes: + + * the "stub to recursive resolver" scenario (also called the "stub + to recursive" scenario in this document) + + * the "recursive resolver to authoritative nameserver" scenario + (also called the "recursive to authoritative" scenario in this + document), and + + * the "nameserver to nameserver" scenario (mainly used for zone + transfers (XFR) [RFC1995] [RFC5936]). + + In other words, this document specifies QUIC as a general-purpose + transport for DNS. + + The specific non-goals of this document are: + + 1. No attempt is made to evade potential blocking of DoQ traffic by + middleboxes. + + 2. No attempt to support server-initiated transactions, which are + used only in DNS Stateful Operations (DSO) [RFC8490]. + + Specifying the transmission of an application over QUIC requires + specifying how the application's messages are mapped to QUIC streams, + and generally how the application will use QUIC. This is done for + HTTP in "Hypertext Transfer Protocol Version 3 (HTTP/3)" [HTTP/3]. + The purpose of this document is to define the way DNS messages can be + transmitted over QUIC. + + DNS over HTTPS (DoH) [RFC8484] can be used with HTTP/3 to get some of + the benefits of QUIC. However, a lightweight direct mapping for DoQ + can be regarded as a more natural fit for both the recursive to + authoritative and zone transfer scenarios, which rarely involve + intermediaries. In these scenarios, the additional overhead of HTTP + is not offset by, for example, benefits of HTTP proxying and caching + behavior. + + In this document, Section 3 presents the reasoning that guided the + proposed design. Section 4 specifies the actual mapping of DoQ. + Section 5 presents guidelines on the implementation, usage, and + deployment of DoQ. + +2. Key Words + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + +3. Design Considerations + + This section and its subsections present the design guidelines that + were used for DoQ. While all other sections in this document are + normative, this section is informative in nature. + +3.1. Provide DNS Privacy + + DoT [RFC7858] defines how to mitigate some of the issues described in + "DNS Privacy Considerations" [RFC9076] by specifying how to transmit + DNS messages over TLS. The "Usage Profiles for DNS over TLS and DNS + over DTLS" [RFC8310] specify Strict and Opportunistic usage profiles + for DoT including how stub resolvers can authenticate recursive + resolvers. + + QUIC connection setup includes the negotiation of security parameters + using TLS, as specified in "Using TLS to Secure QUIC" [RFC9001], + enabling encryption of the QUIC transport. Transmitting DNS messages + over QUIC will provide essentially the same privacy protections as + DoT [RFC7858] including Strict and Opportunistic usage profiles + [RFC8310]. Further discussion on this is provided in Section 7. + +3.2. Design for Minimum Latency + + QUIC is specifically designed to reduce protocol-induced delays, with + features such as: + + 1. Support for 0-RTT data during session resumption. + + 2. Support for advanced packet-loss recovery procedures as specified + in "QUIC Loss Detection and Congestion Control" [RFC9002]. + + 3. Mitigation of head-of-line blocking by allowing parallel delivery + of data on multiple streams. + + This mapping of DNS to QUIC will take advantage of these features in + three ways: + + 1. Optional support for sending 0-RTT data during session resumption + (the security and privacy implications of this are discussed in + later sections). + + 2. Long-lived QUIC connections over which multiple DNS transactions + are performed, generating the sustained traffic required to + benefit from advanced recovery features. + + 3. Mapping of each DNS Query/Response transaction to a separate + stream, to mitigate head-of-line blocking. This enables servers + to respond to queries "out of order". It also enables clients to + process responses as soon as they arrive, without having to wait + for in-order delivery of responses previously posted by the + server. + + These considerations are reflected in the mapping of DNS traffic to + QUIC streams in Section 4.2. + +3.3. Middlebox Considerations + + Using QUIC might allow a protocol to disguise its purpose from + devices on the network path using encryption and traffic analysis + resistance techniques like padding, traffic pacing, and traffic + shaping. This specification does not include any measures that are + designed to avoid such classification; the padding mechanisms defined + in Section 5.4 are intended to obfuscate the specific records + contained in DNS queries and responses, but not the fact that this is + DNS traffic. Consequently, firewalls and other middleboxes might be + able to distinguish DoQ from other protocols that use QUIC, like + HTTP, and apply different treatment. + + The lack of measures in this specification to avoid protocol + classification is not an endorsement of such practices. + +3.4. No Server-Initiated Transactions + + As stated in Section 1, this document does not specify support for + server-initiated transactions within established DoQ connections. + That is, only the initiator of the DoQ connection may send queries + over the connection. + + DSO does support server-initiated transactions within existing + connections. However, DoQ as defined here does not meet the criteria + for an applicable transport for DSO because it does not guarantee in- + order delivery of messages; see Section 4.2 of [RFC8490]. + +4. Specifications + +4.1. Connection Establishment + + DoQ connections are established as described in the QUIC transport + specification [RFC9000]. During connection establishment, DoQ + support is indicated by selecting the Application-Layer Protocol + Negotiation (ALPN) token "doq" in the crypto handshake. + +4.1.1. Port Selection + + By default, a DNS server that supports DoQ MUST listen for and accept + QUIC connections on the dedicated UDP port 853 (Section 8), unless + there is a mutual agreement to use another port. + + By default, a DNS client desiring to use DoQ with a particular server + MUST establish a QUIC connection to UDP port 853 on the server, + unless there is a mutual agreement to use another port. + + DoQ connections MUST NOT use UDP port 53. This recommendation + against use of port 53 for DoQ is to avoid confusion between DoQ and + the use of DNS over UDP [RFC1035]. The risk of confusion exists even + if two parties agreed on port 53, as other parties without knowledge + of that agreement might still try to use that port. + + In the stub to recursive scenario, the use of port 443 as a mutually + agreed alternative port can be operationally beneficial, since port + 443 is used by many services using QUIC and HTTP-3 and is thus less + likely to be blocked than other ports. Several mechanisms for stubs + to discover recursives offering encrypted transports, including the + use of custom ports, are the subject of ongoing work. + +4.2. Stream Mapping and Usage + + The mapping of DNS traffic over QUIC streams takes advantage of the + QUIC stream features detailed in Section 2 of [RFC9000], the QUIC + transport specification. + + DNS query/response traffic [RFC1034] [RFC1035] follows a simple + pattern in which the client sends a query, and the server provides + one or more responses (multiple responses can occur in zone + transfers). + + The mapping specified here requires that the client select a separate + QUIC stream for each query. The server then uses the same stream to + provide all the response messages for that query. In order for + multiple responses to be parsed, a 2-octet length field is used in + exactly the same way as the 2-octet length field defined for DNS over + TCP [RFC1035]. The practical result of this is that the content of + each QUIC stream is exactly the same as the content of a TCP + connection that would manage exactly one query. + + All DNS messages (queries and responses) sent over DoQ connections + MUST be encoded as a 2-octet length field followed by the message + content as specified in [RFC1035]. + + The client MUST select the next available client-initiated + bidirectional stream for each subsequent query on a QUIC connection, + in conformance with the QUIC transport specification [RFC9000]. + Packet losses and other network events might cause queries to arrive + in a different order. Servers SHOULD process queries as they arrive, + as not doing so would cause unnecessary delays. + + The client MUST send the DNS query over the selected stream and MUST + indicate through the STREAM FIN mechanism that no further data will + be sent on that stream. + + The server MUST send the response(s) on the same stream and MUST + indicate, after the last response, through the STREAM FIN mechanism + that no further data will be sent on that stream. + + Therefore, a single DNS transaction consumes a single bidirectional + client-initiated stream. This means that the client's first query + occurs on QUIC stream 0, the second on 4, and so on (see Section 2.1 + of [RFC9000]). + + Servers MAY defer processing of a query until the STREAM FIN has been + indicated on the stream selected by the client. + + Servers and clients MAY monitor the number of "dangling" streams. + These are open streams where the following events have not occurred + after implementation-defined timeouts: + + * the expected queries or responses have not been received or, + + * the expected queries or responses have been received but not the + STREAM FIN + + Implementations MAY impose a limit on the number of such dangling + streams. If limits are encountered, implementations MAY close the + connection. + +4.2.1. DNS Message IDs + + When sending queries over a QUIC connection, the DNS Message ID MUST + be set to 0. The stream mapping for DoQ allows for unambiguous + correlation of queries and responses, so the Message ID field is not + required. + + This has implications for proxying DoQ messages to and from other + transports. For example, proxies may have to manage the fact that + DoQ can support a larger number of outstanding queries on a single + connection than, for example, DNS over TCP, because DoQ is not + limited by the Message ID space. This issue already exists for DoH, + where a Message ID of 0 is recommended. + + When forwarding a DNS message from DoQ over another transport, a DNS + Message ID MUST be generated according to the rules of the protocol + that is in use. When forwarding a DNS message from another transport + over DoQ, the Message ID MUST be set to 0. + +4.3. DoQ Error Codes + + The following error codes are defined for use when abruptly + terminating streams, for use as application protocol error codes when + aborting reading of streams, or for immediately closing connections: + + DOQ_NO_ERROR (0x0): No error. This is used when the connection or + stream needs to be closed, but there is no error to signal. + + DOQ_INTERNAL_ERROR (0x1): The DoQ implementation encountered an + internal error and is incapable of pursuing the transaction or the + connection. + + DOQ_PROTOCOL_ERROR (0x2): The DoQ implementation encountered a + protocol error and is forcibly aborting the connection. + + DOQ_REQUEST_CANCELLED (0x3): A DoQ client uses this to signal that + it wants to cancel an outstanding transaction. + + DOQ_EXCESSIVE_LOAD (0x4): A DoQ implementation uses this to signal + when closing a connection due to excessive load. + + DOQ_UNSPECIFIED_ERROR (0x5): A DoQ implementation uses this in the + absence of a more specific error code. + + DOQ_ERROR_RESERVED (0xd098ea5e): An alternative error code used for + tests. + + See Section 8.4 for details on registering new error codes. + +4.3.1. Transaction Cancellation + + In QUIC, sending STOP_SENDING requests that a peer cease transmission + on a stream. If a DoQ client wishes to cancel an outstanding + request, it MUST issue a QUIC STOP_SENDING, and it SHOULD use the + error code DOQ_REQUEST_CANCELLED. It MAY use a more specific error + code registered according to Section 8.4. The STOP_SENDING request + may be sent at any time but will have no effect if the server + response has already been sent, in which case the client will simply + discard the incoming response. The corresponding DNS transaction + MUST be abandoned. + + Servers that receive STOP_SENDING act in accordance with Section 3.5 + of [RFC9000]. Servers SHOULD NOT continue processing a DNS + transaction if they receive a STOP_SENDING. + + Servers MAY impose implementation limits on the total number or rate + of cancellation requests. If limits are encountered, servers MAY + close the connection. In this case, servers wanting to help client + debugging MAY use the error code DOQ_EXCESSIVE_LOAD. There is always + a trade-off between helping good faith clients debug issues and + allowing denial-of-service attackers to test server defenses; + depending on circumstances servers might very well choose to send + different error codes. + + Note that this mechanism provides a way for secondaries to cancel a + single zone transfer occurring on a given stream without having to + close the QUIC connection. + + Servers MUST NOT continue processing a DNS transaction if they + receive a RESET_STREAM request from the client before the client + indicates the STREAM FIN. The server MUST issue a RESET_STREAM to + indicate that the transaction is abandoned unless: + + * it has already done so for another reason or + + * it has already both sent the response and indicated the STREAM + FIN. + +4.3.2. Transaction Errors + + Servers normally complete transactions by sending a DNS response (or + responses) on the transaction's stream, including cases where the DNS + response indicates a DNS error. For example, a client SHOULD be + notified of a Server Failure (SERVFAIL, [RFC1035]) through a response + with the Response Code set to SERVFAIL. + + If a server is incapable of sending a DNS response due to an internal + error, it SHOULD issue a QUIC RESET_STREAM frame. The error code + SHOULD be set to DOQ_INTERNAL_ERROR. The corresponding DNS + transaction MUST be abandoned. Clients MAY limit the number of + unsolicited QUIC RESET_STREAM frames received on a connection before + choosing to close the connection. + + Note that this mechanism provides a way for primaries to abort a + single zone transfer occurring on a given stream without having to + close the QUIC connection. + +4.3.3. Protocol Errors + + Other error scenarios can occur due to malformed, incomplete, or + unexpected messages during a transaction. These include (but are not + limited to): + + * a client or server receives a message with a non-zero Message ID + + * a client or server receives a STREAM FIN before receiving all the + bytes for a message indicated in the 2-octet length field + + * a client receives a STREAM FIN before receiving all the expected + responses + + * a server receives more than one query on a stream + + * a client receives a different number of responses on a stream than + expected (e.g., multiple responses to a query for an A record) + + * a client receives a STOP_SENDING request + + * the client or server does not indicate the expected STREAM FIN + after sending requests or responses (see Section 4.2) + + * an implementation receives a message containing the edns-tcp- + keepalive EDNS(0) Option [RFC7828] (see Section 5.5.2) + + * a client or a server attempts to open a unidirectional QUIC stream + + * a server attempts to open a server-initiated bidirectional QUIC + stream + + * a server receives a "replayable" transaction in 0-RTT data (for + servers not willing to handle this case, see Section 4.5) + + If a peer encounters such an error condition, it is considered a + fatal error. It SHOULD forcibly abort the connection using QUIC's + CONNECTION_CLOSE mechanism and SHOULD use the DoQ error code + DOQ_PROTOCOL_ERROR. In some cases, it MAY instead silently abandon + the connection, which uses fewer of the local resources but makes + debugging at the offending node more difficult. + + It is noted that the restrictions on use of the above EDNS(0) option + has implications for proxying messages from TCP/DoT/DoH over DoQ. + +4.3.4. Alternative Error Codes + + This specification describes specific error codes in Sections 4.3.1, + 4.3.2, and 4.3.3. These error codes are meant to facilitate + investigation of failures and other incidents. New error codes may + be defined in future versions of DoQ or registered as specified in + Section 8.4. + + Because new error codes can be defined without negotiation, use of an + error code in an unexpected context or receipt of an unknown error + code MUST be treated as equivalent to DOQ_UNSPECIFIED_ERROR. + + Implementations MAY wish to test the support for the error code + extension mechanism by using error codes not listed in this document, + or they MAY use DOQ_ERROR_RESERVED. + +4.4. Connection Management + + Section 10 of [RFC9000], the QUIC transport specification, specifies + that connections can be closed in three ways: + + * idle timeout + + * immediate close + + * stateless reset + + Clients and servers implementing DoQ SHOULD negotiate use of the idle + timeout. Closing on idle timeout is done without any packet + exchange, which minimizes protocol overhead. Per Section 10.1 of + [RFC9000], the QUIC transport specification, the effective value of + the idle timeout is computed as the minimum of the values advertised + by the two endpoints. Practical considerations on setting the idle + timeout are discussed in Section 5.5.2. + + Clients SHOULD monitor the idle time incurred on their connection to + the server, defined by the time spent since the last packet from the + server has been received. When a client prepares to send a new DNS + query to the server, it SHOULD check whether the idle time is + sufficiently lower than the idle timer. If it is, the client SHOULD + send the DNS query over the existing connection. If not, the client + SHOULD establish a new connection and send the query over that + connection. + + Clients MAY discard their connections to the server before the idle + timeout expires. A client that has outstanding queries SHOULD close + the connection explicitly using QUIC's CONNECTION_CLOSE mechanism and + the DoQ error code DOQ_NO_ERROR. + + Clients and servers MAY close the connection for a variety of other + reasons, indicated using QUIC's CONNECTION_CLOSE. Client and servers + that send packets over a connection discarded by their peer might + receive a stateless reset indication. If a connection fails, all the + in-progress transactions on that connection MUST be abandoned. + +4.5. Session Resumption and 0-RTT + + A client MAY take advantage of the session resumption and 0-RTT + mechanisms supported by QUIC transport [RFC9000] and QUIC TLS + [RFC9001] if the server supports them. Clients SHOULD consider + potential privacy issues associated with session resumption before + deciding to use this mechanism and specifically evaluate the trade- + offs presented in the various sections of this document. The privacy + issues are detailed in Sections 7.1 and 7.2, and the implementation + considerations are discussed in Section 5.5.3. + + The 0-RTT mechanism MUST NOT be used to send DNS requests that are + not "replayable" transactions. In this specification, only + transactions that have an OPCODE of QUERY or NOTIFY are considered + replayable; therefore, other OPCODES MUST NOT be sent in 0-RTT data. + See Appendix A for a detailed discussion of why NOTIFY is included + here. + + Servers MAY support session resumption, and MAY do that with or + without supporting 0-RTT, using the mechanisms described in + Section 4.6.1 of [RFC9001]. Servers supporting 0-RTT MUST NOT + immediately process non-replayable transactions received in 0-RTT + data but instead MUST adopt one of the following behaviors: + + * Queue the offending transaction and only process it after the QUIC + handshake has been completed, as defined in Section 4.1.1 of + [RFC9001]. + + * Reply to the offending transaction with a response code REFUSED + and an Extended DNS Error Code (EDE) "Too Early" using the + extended RCODE mechanisms defined in [RFC6891] and the extended + DNS errors defined in [RFC8914]; see Section 8.3. + + * Close the connection with the error code DOQ_PROTOCOL_ERROR. + +4.6. Message Sizes + + DoQ queries and responses are sent on QUIC streams, which in theory + can carry up to 2^62 bytes. However, DNS messages are restricted in + practice to a maximum size of 65535 bytes. This maximum size is + enforced by the use of a 2-octet message length field in DNS over TCP + [RFC1035] and DoT [RFC7858], and by the definition of the + "application/dns-message" for DoH [RFC8484]. DoQ enforces the same + restriction. + + The Extension Mechanisms for DNS (EDNS(0)) [RFC6891] allow peers to + specify the UDP message size. This parameter is ignored by DoQ. DoQ + implementations always assume that the maximum message size is 65535 + bytes. + +5. Implementation Requirements + +5.1. Authentication + + For the stub to recursive scenario, the authentication requirements + are the same as described in DoT [RFC7858] and "Usage Profiles for + DNS over TLS and DNS over DTLS" [RFC8310]. [RFC8932] states that DNS + privacy services SHOULD provide credentials that clients can use to + authenticate the server. Given this, and to align with the + authentication model for DoH, DoQ stubs SHOULD use a Strict usage + profile. Client authentication for the encrypted stub to recursive + scenario is not described in any DNS RFC. + + For zone transfer, the authentication requirements are the same as + described in [RFC9103]. + + For the recursive to authoritative scenario, authentication + requirements are unspecified at the time of writing and are the + subject of ongoing work in the DPRIVE WG. + +5.2. Fallback to Other Protocols on Connection Failure + + If the establishment of the DoQ connection fails, clients MAY attempt + to fall back to DoT and then potentially cleartext, as specified in + DoT [RFC7858] and "Usage Profiles for DNS over TLS and DNS over DTLS" + [RFC8310], depending on their usage profile. + + DNS clients SHOULD remember server IP addresses that don't support + DoQ. Mobile clients might also remember the lack of DoQ support by + given IP addresses on a per-context basis (e.g., per network or + provisioning domain). + + Timeouts, connection refusals, and QUIC handshake failures are + indicators that a server does not support DoQ. Clients SHOULD NOT + attempt DoQ queries to a server that does not support DoQ for a + reasonable period (such as one hour per server). DNS clients + following an out-of-band key-pinned usage profile [RFC7858] MAY be + more aggressive about retrying after DoQ connection failures. + +5.3. Address Validation + + Section 8 of [RFC9000], the QUIC transport specification, defines + Address Validation procedures to avoid servers being used in address + amplification attacks. DoQ implementations MUST conform to this + specification, which limits the worst-case amplification to a factor + 3. + + DoQ implementations SHOULD consider configuring servers to use the + Address Validation using Retry Packets procedure defined in + Section 8.1.2 of [RFC9000], the QUIC transport specification. This + procedure imposes a 1-RTT delay for verifying the return routability + of the source address of a client, similar to the DNS Cookies + mechanism [RFC7873]. + + DoQ implementations that configure Address Validation using Retry + Packets SHOULD implement the Address Validation for Future + Connections procedure defined in Section 8.1.3 of [RFC9000], the QUIC + transport specification. This defines how servers can send NEW_TOKEN + frames to clients after the client address is validated in order to + avoid the 1-RTT penalty during subsequent connections by the client + from the same address. + +5.4. Padding + + Implementations MUST protect against the traffic analysis attacks + described in Section 7.5 by the judicious injection of padding. This + could be done either by padding individual DNS messages using the + EDNS(0) Padding Option [RFC7830] or by padding QUIC packets (see + Section 19.1 of [RFC9000]). + + In theory, padding at the QUIC packet level could result in better + performance for the equivalent protection, because the amount of + padding can take into account non-DNS frames such as acknowledgements + or flow control updates, and also because QUIC packets can carry + multiple DNS messages. However, applications can only control the + amount of padding in QUIC packets if the implementation of QUIC + exposes adequate APIs. This leads to the following recommendations: + + * If the implementation of QUIC exposes APIs to set a padding + policy, DoQ SHOULD use that API to align the packet length to a + small set of fixed sizes. + + * If padding at the QUIC packet level is not available or not used, + DoQ MUST ensure that all DNS queries and responses are padded to a + small set of fixed sizes, using the EDNS(0) padding extension as + specified in [RFC7830]. + + Implementations might choose not to use a QUIC API for padding if it + is significantly simpler to reuse existing DNS message padding logic + that is applied to other encrypted transports. + + In the absence of a standard policy for padding sizes, + implementations SHOULD follow the recommendations of the Experimental + status "Padding Policies for Extension Mechanisms for DNS (EDNS(0))" + [RFC8467]. While Experimental, these recommendations are referenced + because they are implemented and deployed for DoT and provide a way + for implementations to be fully compliant with this specification. + +5.5. Connection Handling + + "DNS Transport over TCP - Implementation Requirements" [RFC7766] + provides updated guidance on DNS over TCP, some of which is + applicable to DoQ. This section provides similar advice on + connection handling for DoQ. + +5.5.1. Connection Reuse + + Historic implementations of DNS clients are known to open and close + TCP connections for each DNS query. To amortize connection setup + costs, both clients and servers SHOULD support connection reuse by + sending multiple queries and responses over a single persistent QUIC + connection. + + In order to achieve performance on par with UDP, DNS clients SHOULD + send their queries concurrently over the QUIC streams on a QUIC + connection. That is, when a DNS client sends multiple queries to a + server over a QUIC connection, it SHOULD NOT wait for an outstanding + reply before sending the next query. + +5.5.2. Resource Management + + Proper management of established and idle connections is important to + the healthy operation of a DNS server. + + An implementation of DoQ SHOULD follow best practices similar to + those specified for DNS over TCP [RFC7766], in particular with regard + to: + + * Concurrent Connections (Section 6.2.2 of [RFC7766], updated by + Section 6.4 of [RFC9103]) + + * Security Considerations (Section 10 of [RFC7766]) + + Failure to do so may lead to resource exhaustion and denial of + service. + + Clients that want to maintain long duration DoQ connections SHOULD + use the idle timeout mechanisms defined in Section 10.1 of [RFC9000], + the QUIC transport specification. Clients and servers MUST NOT send + the edns-tcp-keepalive EDNS(0) Option [RFC7828] in any messages sent + on a DoQ connection (because it is specific to the use of TCP/TLS as + a transport). + + This document does not make specific recommendations for timeout + values on idle connections. Clients and servers should reuse and/or + close connections depending on the level of available resources. + Timeouts may be longer during periods of low activity and shorter + during periods of high activity. + +5.5.3. Using 0-RTT and Session Resumption + + Using 0-RTT for DoQ has many compelling advantages. Clients can + establish connections and send queries without incurring a connection + delay. Servers can thus negotiate low values of the connection + timers, which reduces the total number of connections that they need + to manage. They can do that because the clients that use 0-RTT will + not incur latency penalties if new connections are required for a + query. + + Session resumption and 0-RTT data transmission create privacy risks + detailed in Sections 7.1 and 7.2. The following recommendations are + meant to reduce the privacy risks while enjoying the performance + benefits of 0-RTT data, subject to the restrictions specified in + Section 4.5. + + Clients SHOULD use resumption tickets only once, as specified in + Appendix C.4 of [RFC8446]. By default, clients SHOULD NOT use + session resumption if the client's connectivity has changed. + + Clients could receive address validation tokens from the server using + the NEW_TOKEN mechanism; see Section 8 of [RFC9000]. The associated + tracking risks are mentioned in Section 7.3. Clients SHOULD only use + the address validation tokens when they are also using session + resumption thus avoiding additional tracking risks. + + Servers SHOULD issue session resumption tickets with a sufficiently + long lifetime (e.g., 6 hours), so that clients are not tempted to + either keep the connection alive or frequently poll the server to + renew session resumption tickets. Servers SHOULD implement the anti- + replay mechanisms specified in Section 8 of [RFC8446]. + +5.5.4. Controlling Connection Migration for Privacy + + DoQ implementations might consider using the connection migration + features defined in Section 9 of [RFC9000]. These features enable + connections to continue operating as the client's connectivity + changes. As detailed in Section 7.4, these features trade off + privacy for latency. By default, clients SHOULD be configured to + prioritize privacy and start new sessions if their connectivity + changes. + +5.6. Processing Queries in Parallel + + As specified in Section 7 of [RFC7766] "DNS Transport over TCP - + Implementation Requirements", resolvers are RECOMMENDED to support + the preparing of responses in parallel and sending them out of order. + In DoQ, they do that by sending responses on their specific stream as + soon as possible, without waiting for availability of responses for + previously opened streams. + +5.7. Zone Transfer + + [RFC9103] specifies zone transfer over TLS (XoT) and includes updates + to [RFC1995] (IXFR), [RFC5936] (AXFR), and [RFC7766]. Considerations + relating to the reuse of XoT connections described there apply + analogously to zone transfers performed using DoQ connections. One + reason for reiterating such specific guidance is the lack of + effective connection reuse in existing TCP/TLS zone transfer + implementations today. The following recommendations apply: + + * DoQ servers MUST be able to handle multiple concurrent IXFR + requests on a single QUIC connection. + + * DoQ servers MUST be able to handle multiple concurrent AXFR + requests on a single QUIC connection. + + * DoQ implementations SHOULD + + - use the same QUIC connection for both AXFR and IXFR requests to + the same primary + + - send those requests in parallel as soon as they are queued, + i.e., do not wait for a response before sending the next query + on the connection (this is analogous to pipelining requests on + a TCP/TLS connection) + + - send the response(s) for each request as soon as they are + available, i.e., response streams MAY be sent intermingled + +5.8. Flow Control Mechanisms + + Servers and clients manage flow control using the mechanisms defined + in Section 4 of [RFC9000]. These mechanisms allow clients and + servers to specify how many streams can be created, how much data can + be sent on a stream, and how much data can be sent on the union of + all streams. For DoQ, controlling how many streams are created + allows servers to control how many new requests the client can send + on a given connection. + + Flow control exists to protect endpoint resources. For servers, + global and per-stream flow control limits control how much data can + be sent by clients. The same mechanisms allow clients to control how + much data can be sent by servers. Values that are too small will + unnecessarily limit performance. Values that are too large might + expose endpoints to overload or memory exhaustion. Implementations + or deployments will need to adjust flow control limits to balance + these concerns. In particular, zone transfer implementations will + need to control these limits carefully to ensure both large and + concurrent zone transfers are well managed. + + Initial values of parameters control how many requests and how much + data can be sent by clients and servers at the beginning of the + connection. These values are specified in transport parameters + exchanged during the connection handshake. The parameter values + received in the initial connection also control how many requests and + how much data can be sent by clients using 0-RTT data in a resumed + connection. Using too small values of these initial parameters would + restrict the usefulness of allowing 0-RTT data. + +6. Security Considerations + + A Threat Analysis of the Domain Name System is found in [RFC3833]. + This analysis was written before the development of DoT, DoH, and + DoQ, and probably needs to be updated. + + The security considerations of DoQ should be comparable to those of + DoT [RFC7858]. DoT as specified in [RFC7858] only addresses the stub + to recursive scenario, but the considerations about person-in-the- + middle attacks, middleboxes, and caching of data from cleartext + connections also apply for DoQ to the resolver to authoritative + server scenario. As stated in Section 5.1, the authentication + requirements for securing zone transfer using DoQ are the same as + those for zone transfer over DoT; therefore, the general security + considerations are entirely analogous to those described in + [RFC9103]. + + DoQ relies on QUIC, which itself relies on TLS 1.3 and thus supports + by default the protections against downgrade attacks described in + [BCP195]. QUIC-specific issues and their mitigations are described + in Section 21 of [RFC9000]. + +7. Privacy Considerations + + The general considerations of encrypted transports provided in "DNS + Privacy Considerations" [RFC9076] apply to DoQ. The specific + considerations provided there do not differ between DoT and DoQ, and + they are not discussed further here. Similarly, "Recommendations for + DNS Privacy Service Operators" [RFC8932] (which covers operational, + policy, and security considerations for DNS privacy services) is also + applicable to DoQ services. + + QUIC incorporates the mechanisms of TLS 1.3 [RFC8446], and this + enables QUIC transmission of "0-RTT" data. This can provide + interesting latency gains, but it raises two concerns: + + 1. Adversaries could replay the 0-RTT data and infer its content + from the behavior of the receiving server. + + 2. The 0-RTT mechanism relies on TLS session resumption, which can + provide linkability between successive client sessions. + + These issues are developed in Sections 7.1 and 7.2. + +7.1. Privacy Issues with 0-RTT data + + The 0-RTT data can be replayed by adversaries. That data may trigger + queries by a recursive resolver to authoritative resolvers. + Adversaries may be able to pick a time at which the recursive + resolver outgoing traffic is observable and thus find out what name + was queried for in the 0-RTT data. + + This risk is in fact a subset of the general problem of observing the + behavior of the recursive resolver discussed in "DNS Privacy + Considerations" [RFC9076]. The attack is partially mitigated by + reducing the observability of this traffic. The mandatory replay + protection mechanisms in TLS 1.3 [RFC8446] limit but do not eliminate + the risk of replay. 0-RTT packets can only be replayed within a + narrow window, which is only wide enough to account for variations in + clock skew and network transmission. + + The recommendation for TLS 1.3 [RFC8446] is that the capability to + use 0-RTT data should be turned off by default and only enabled if + the user clearly understands the associated risks. In the case of + DoQ, allowing 0-RTT data provides significant performance gains, and + there is a concern that a recommendation to not use it would simply + be ignored. Instead, a set of practical recommendations is provided + in Sections 4.5 and 5.5.3. + + The specifications in Section 4.5 block the most obvious risks of + replay attacks, as they only allow for transactions that will not + change the long-term state of the server. + + The attacks described above apply to the stub resolver to recursive + resolver scenario, but similar attacks might be envisaged in the + recursive resolver to authoritative resolver scenario, and the same + mitigations apply. + +7.2. Privacy Issues with Session Resumption + + The QUIC session resumption mechanism reduces the cost of re- + establishing sessions and enables 0-RTT data. There is a linkability + issue associated with session resumption, if the same resumption + token is used several times. Attackers on path between client and + server could observe repeated usage of the token and use that to + track the client over time or over multiple locations. + + The session resumption mechanism allows servers to correlate the + resumed sessions with the initial sessions and thus to track the + client. This creates a virtual long duration session. The series of + queries in that session can be used by the server to identify the + client. Servers can most probably do that already if the client + address remains constant, but session resumption tickets also enable + tracking after changes of the client's address. + + The recommendations in Section 5.5.3 are designed to mitigate these + risks. Using session tickets only once mitigates the risk of + tracking by third parties. Refusing to resume a session if addresses + change mitigates the incremental risk of tracking by the server (but + the risk of tracking by IP address remains). + + The privacy trade-offs here may be context specific. Stub resolvers + will have a strong motivation to prefer privacy over latency since + they often change location. However, recursive resolvers that use a + small set of static IP addresses are more likely to prefer the + reduced latency provided by session resumption and may consider this + a valid reason to use resumption tickets even if the IP address + changed between sessions. + + Encrypted zone transfer ([RFC9103]) explicitly does not attempt to + hide the identity of the parties involved in the transfer; at the + same time, such transfers are not particularly latency sensitive. + This means that applications supporting zone transfers may decide to + apply the same protections as stub to recursive applications. + +7.3. Privacy Issues with Address Validation Tokens + + QUIC specifies address validation mechanisms in Section 8 of + [RFC9000]. Use of an address validation token allows QUIC servers to + avoid an extra RTT for new connections. Address validation tokens + are typically tied to an IP address. QUIC clients normally only use + these tokens when setting up a new connection from a previously used + address. However, clients are not always aware that they are using a + new address. This could be due to NAT, or because the client does + not have an API available to check if the IP address has changed + (which can be quite often for IPv6). There is a linkability risk if + clients mistakenly use address validation tokens after unknowingly + moving to a new location. + + The recommendations in Section 5.5.3 mitigates this risk by tying the + usage of the NEW_TOKEN to that of session resumption, though this + recommendation does not cover the case where the client is unaware of + the address change. + +7.4. Privacy Issues with Long Duration Sessions + + A potential alternative to session resumption is the use of long + duration sessions: if a session remains open for a long time, new + queries can be sent without incurring connection establishment + delays. It is worth pointing out that the two solutions have similar + privacy characteristics. Session resumption may allow servers to + keep track of the IP addresses of clients, but long duration sessions + have the same effect. + + In particular, a DoQ implementation might take advantage of the + connection migration features of QUIC to maintain a session even if + the client's connectivity changes, for example, if the client + migrates from a Wi-Fi connection to a cellular network connection and + then to another Wi-Fi connection. The server would be able to track + the client location by monitoring the succession of IP addresses used + by the long duration connection. + + The recommendation in Section 5.5.4 mitigates the privacy concerns + related to long duration sessions using multiple client addresses. + +7.5. Traffic Analysis + + Even though QUIC packets are encrypted, adversaries can gain + information from observing packet lengths, in both queries and + responses, as well as packet timing. Many DNS requests are emitted + by web browsers. Loading a specific web page may require resolving + dozens of DNS names. If an application adopts a simple mapping of + one query or response per packet, or "one QUIC STREAM frame per + packet", then the succession of packet lengths may provide enough + information to identify the requested site. + + Implementations SHOULD use the mechanisms defined in Section 5.4 to + mitigate this attack. + +8. IANA Considerations + +8.1. Registration of a DoQ Identification String + + This document creates a new registration for the identification of + DoQ in the "TLS Application-Layer Protocol Negotiation (ALPN) + Protocol IDs" registry [RFC7301]. + + The "doq" string identifies DoQ: + + Protocol: DoQ + + Identification Sequence: 0x64 0x6F 0x71 ("doq") + + Specification: This document + +8.2. Reservation of a Dedicated Port + + For both TCP and UDP, port 853 is currently reserved for "DNS query- + response protocol run over TLS/DTLS" [RFC7858]. + + However, the specification for DNS over DTLS (DoD) [RFC8094] is + experimental, limited to stub to resolver, and no implementations or + deployments currently exist to the authors' knowledge (even though + several years have passed since the specification was published). + + This specification additionally reserves the use of UDP port 853 for + DoQ. QUIC version 1 was designed to be able to coexist with other + protocols on the same port, including DTLS; see Section 17.2 of + [RFC9000]. This means that deployments that serve DoD and DoQ (QUIC + version 1) on the same port will be able to demultiplex the two due + to the second most significant bit in each UDP payload. Such + deployments ought to check the signatures of future versions or + extensions (e.g., [GREASING-QUIC]) of QUIC and DTLS before deploying + them to serve DNS on the same port. + + IANA has updated the following value in the "Service Name and + Transport Protocol Port Number Registry" in the System range. The + registry for that range requires IETF Review or IESG Approval + [RFC6335]. + + Service Name: domain-s + + Port Number: 853 + + Transport Protocol(s): UDP + + Assignee: IESG + + Contact: IETF Chair + + Description: DNS query-response protocol run over DTLS or QUIC + + Reference: [RFC7858][RFC8094] This document + + Additionally, IANA has updated the Description field for the + corresponding TCP port 853 allocation to be "DNS query-response + protocol run over TLS" and removed [RFC8094] from the TCP + allocation's Reference field for consistency and clarity. + +8.3. Reservation of an Extended DNS Error Code: Too Early + + IANA has registered the following value in the "Extended DNS Error + Codes" registry [RFC8914]: + + INFO-CODE: 26 + + Purpose: Too Early + + Reference: This document + +8.4. DNS-over-QUIC Error Codes Registry + + IANA has added a registry for "DNS-over-QUIC Error Codes" on the + "Domain Name System (DNS) Parameters" web page. + + The "DNS-over-QUIC Error Codes" registry governs a 62-bit space. + This space is split into three regions that are governed by different + policies: + + * Permanent registrations for values between 0x00 and 0x3f (in + hexadecimal; inclusive), which are assigned using Standards Action + or IESG Approval as defined in Sections 4.9 and 4.10 of [RFC8126] + + * Permanent registrations for values larger than 0x3f, which are + assigned using the Specification Required policy ([RFC8126]) + + * Provisional registrations for values larger than 0x3f, which + require Expert Review, as defined in Section 4.5 of [RFC8126]. + + Provisional reservations share the range of values larger than 0x3f + with some permanent registrations. This is by design to enable + conversion of provisional registrations into permanent registrations + without requiring changes in deployed systems. (This design is + aligned with the principles set in Section 22 of [RFC9000].) + + Registrations in this registry MUST include the following fields: + + Value: The assigned codepoint + + Status: "Permanent" or "Provisional" + + Contact: Contact details for the registrant + + In addition, permanent registrations MUST include: + + Error: A short mnemonic for the parameter + + Specification: A reference to a publicly available specification for + the value (optional for provisional registrations) + + Description: A brief description of the error code semantics, which + MAY be a summary if a specification reference is provided + + Provisional registrations of codepoints are intended to allow for + private use and experimentation with extensions to DoQ. However, + provisional registrations could be reclaimed and reassigned for other + purposes. In addition to the parameters listed above, provisional + registrations MUST include: + + Date: The date of last update to the registration + + A request to update the date on any provisional registration can be + made without review from the designated expert(s). + + The initial content of this registry is shown in Table 1 and all + entries share the following fields: + + Status: Permanent + + Contact: DPRIVE WG + + Specification: Section 4.3 + + +============+=======================+=============================+ + | Value | Error | Description | + +============+=======================+=============================+ + | 0x0 | DOQ_NO_ERROR | No error | + +------------+-----------------------+-----------------------------+ + | 0x1 | DOQ_INTERNAL_ERROR | Implementation error | + +------------+-----------------------+-----------------------------+ + | 0x2 | DOQ_PROTOCOL_ERROR | Generic protocol violation | + +------------+-----------------------+-----------------------------+ + | 0x3 | DOQ_REQUEST_CANCELLED | Request cancelled by client | + +------------+-----------------------+-----------------------------+ + | 0x4 | DOQ_EXCESSIVE_LOAD | Closing a connection for | + | | | excessive load | + +------------+-----------------------+-----------------------------+ + | 0x5 | DOQ_UNSPECIFIED_ERROR | No error reason specified | + +------------+-----------------------+-----------------------------+ + | 0xd098ea5e | DOQ_ERROR_RESERVED | Alternative error code used | + | | | for tests | + +------------+-----------------------+-----------------------------+ + + Table 1: Initial DNS-over-QUIC Error Codes Entries + +9. References + +9.1. Normative References + + [RFC1034] Mockapetris, P., "Domain names - concepts and facilities", + STD 13, RFC 1034, DOI 10.17487/RFC1034, November 1987, + <https://www.rfc-editor.org/info/rfc1034>. + + [RFC1035] Mockapetris, P., "Domain names - implementation and + specification", STD 13, RFC 1035, DOI 10.17487/RFC1035, + November 1987, <https://www.rfc-editor.org/info/rfc1035>. + + [RFC1995] Ohta, M., "Incremental Zone Transfer in DNS", RFC 1995, + DOI 10.17487/RFC1995, August 1996, + <https://www.rfc-editor.org/info/rfc1995>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC5936] Lewis, E. and A. Hoenes, Ed., "DNS Zone Transfer Protocol + (AXFR)", RFC 5936, DOI 10.17487/RFC5936, June 2010, + <https://www.rfc-editor.org/info/rfc5936>. + + [RFC6891] Damas, J., Graff, M., and P. Vixie, "Extension Mechanisms + for DNS (EDNS(0))", STD 75, RFC 6891, + DOI 10.17487/RFC6891, April 2013, + <https://www.rfc-editor.org/info/rfc6891>. + + [RFC7301] Friedl, S., Popov, A., Langley, A., and E. Stephan, + "Transport Layer Security (TLS) Application-Layer Protocol + Negotiation Extension", RFC 7301, DOI 10.17487/RFC7301, + July 2014, <https://www.rfc-editor.org/info/rfc7301>. + + [RFC7766] Dickinson, J., Dickinson, S., Bellis, R., Mankin, A., and + D. Wessels, "DNS Transport over TCP - Implementation + Requirements", RFC 7766, DOI 10.17487/RFC7766, March 2016, + <https://www.rfc-editor.org/info/rfc7766>. + + [RFC7830] Mayrhofer, A., "The EDNS(0) Padding Option", RFC 7830, + DOI 10.17487/RFC7830, May 2016, + <https://www.rfc-editor.org/info/rfc7830>. + + [RFC7858] Hu, Z., Zhu, L., Heidemann, J., Mankin, A., Wessels, D., + and P. Hoffman, "Specification for DNS over Transport + Layer Security (TLS)", RFC 7858, DOI 10.17487/RFC7858, May + 2016, <https://www.rfc-editor.org/info/rfc7858>. + + [RFC8126] Cotton, M., Leiba, B., and T. Narten, "Guidelines for + Writing an IANA Considerations Section in RFCs", BCP 26, + RFC 8126, DOI 10.17487/RFC8126, June 2017, + <https://www.rfc-editor.org/info/rfc8126>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + + [RFC8310] Dickinson, S., Gillmor, D., and T. Reddy, "Usage Profiles + for DNS over TLS and DNS over DTLS", RFC 8310, + DOI 10.17487/RFC8310, March 2018, + <https://www.rfc-editor.org/info/rfc8310>. + + [RFC8446] Rescorla, E., "The Transport Layer Security (TLS) Protocol + Version 1.3", RFC 8446, DOI 10.17487/RFC8446, August 2018, + <https://www.rfc-editor.org/info/rfc8446>. + + [RFC8467] Mayrhofer, A., "Padding Policies for Extension Mechanisms + for DNS (EDNS(0))", RFC 8467, DOI 10.17487/RFC8467, + October 2018, <https://www.rfc-editor.org/info/rfc8467>. + + [RFC8914] Kumari, W., Hunt, E., Arends, R., Hardaker, W., and D. + Lawrence, "Extended DNS Errors", RFC 8914, + DOI 10.17487/RFC8914, October 2020, + <https://www.rfc-editor.org/info/rfc8914>. + + [RFC9000] Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [RFC9001] Thomson, M., Ed. and S. Turner, Ed., "Using TLS to Secure + QUIC", RFC 9001, DOI 10.17487/RFC9001, May 2021, + <https://www.rfc-editor.org/info/rfc9001>. + + [RFC9103] Toorop, W., Dickinson, S., Sahib, S., Aras, P., and A. + Mankin, "DNS Zone Transfer over TLS", RFC 9103, + DOI 10.17487/RFC9103, August 2021, + <https://www.rfc-editor.org/info/rfc9103>. + +9.2. Informative References + + [BCP195] Sheffer, Y., Holz, R., and P. Saint-Andre, + "Recommendations for Secure Use of Transport Layer + Security (TLS) and Datagram Transport Layer Security + (DTLS)", BCP 195, RFC 7525, May 2015. + + Moriarty, K. and S. Farrell, "Deprecating TLS 1.0 and TLS + 1.1", BCP 195, RFC 8996, March 2021. + + <https://www.rfc-editor.org/info/bcp195> + + [DNS-TERMS] + Hoffman, P. and K. Fujiwara, "DNS Terminology", Work in + Progress, Internet-Draft, draft-ietf-dnsop-rfc8499bis-03, + 28 September 2021, <https://datatracker.ietf.org/doc/html/ + draft-ietf-dnsop-rfc8499bis-03>. + + [DNS0RTT] Kahn Gillmor, D., "DNS + 0-RTT", Message to DNS-Privacy WG + mailing list, 6 April 2016, <https://www.ietf.org/mail- + archive/web/dns-privacy/current/msg01276.html>. + + [GREASING-QUIC] + Thomson, M., "Greasing the QUIC Bit", Work in Progress, + Internet-Draft, draft-ietf-quic-bit-grease-02, 10 November + 2021, <https://datatracker.ietf.org/doc/html/draft-ietf- + quic-bit-grease-02>. + + [HTTP/3] Bishop, M., Ed., "Hypertext Transfer Protocol Version 3 + (HTTP/3)", Work in Progress, Internet-Draft, draft-ietf- + quic-http-34, 2 February 2021, + <https://datatracker.ietf.org/doc/html/draft-ietf-quic- + http-34>. + + [RFC1996] Vixie, P., "A Mechanism for Prompt Notification of Zone + Changes (DNS NOTIFY)", RFC 1996, DOI 10.17487/RFC1996, + August 1996, <https://www.rfc-editor.org/info/rfc1996>. + + [RFC3833] Atkins, D. and R. Austein, "Threat Analysis of the Domain + Name System (DNS)", RFC 3833, DOI 10.17487/RFC3833, August + 2004, <https://www.rfc-editor.org/info/rfc3833>. + + [RFC6335] Cotton, M., Eggert, L., Touch, J., Westerlund, M., and S. + Cheshire, "Internet Assigned Numbers Authority (IANA) + Procedures for the Management of the Service Name and + Transport Protocol Port Number Registry", BCP 165, + RFC 6335, DOI 10.17487/RFC6335, August 2011, + <https://www.rfc-editor.org/info/rfc6335>. + + [RFC7828] Wouters, P., Abley, J., Dickinson, S., and R. Bellis, "The + edns-tcp-keepalive EDNS0 Option", RFC 7828, + DOI 10.17487/RFC7828, April 2016, + <https://www.rfc-editor.org/info/rfc7828>. + + [RFC7873] Eastlake 3rd, D. and M. Andrews, "Domain Name System (DNS) + Cookies", RFC 7873, DOI 10.17487/RFC7873, May 2016, + <https://www.rfc-editor.org/info/rfc7873>. + + [RFC8094] Reddy, T., Wing, D., and P. Patil, "DNS over Datagram + Transport Layer Security (DTLS)", RFC 8094, + DOI 10.17487/RFC8094, February 2017, + <https://www.rfc-editor.org/info/rfc8094>. + + [RFC8484] Hoffman, P. and P. McManus, "DNS Queries over HTTPS + (DoH)", RFC 8484, DOI 10.17487/RFC8484, October 2018, + <https://www.rfc-editor.org/info/rfc8484>. + + [RFC8490] Bellis, R., Cheshire, S., Dickinson, J., Dickinson, S., + Lemon, T., and T. Pusateri, "DNS Stateful Operations", + RFC 8490, DOI 10.17487/RFC8490, March 2019, + <https://www.rfc-editor.org/info/rfc8490>. + + [RFC8932] Dickinson, S., Overeinder, B., van Rijswijk-Deij, R., and + A. Mankin, "Recommendations for DNS Privacy Service + Operators", BCP 232, RFC 8932, DOI 10.17487/RFC8932, + October 2020, <https://www.rfc-editor.org/info/rfc8932>. + + [RFC9002] Iyengar, J., Ed. and I. Swett, Ed., "QUIC Loss Detection + and Congestion Control", RFC 9002, DOI 10.17487/RFC9002, + May 2021, <https://www.rfc-editor.org/info/rfc9002>. + + [RFC9076] Wicinski, T., Ed., "DNS Privacy Considerations", RFC 9076, + DOI 10.17487/RFC9076, July 2021, + <https://www.rfc-editor.org/info/rfc9076>. + +Appendix A. The NOTIFY Service + + This appendix discusses why it is considered acceptable to send + NOTIFY (see [RFC1996]) in 0-RTT data. + + Section 4.5 says "The 0-RTT mechanism MUST NOT be used to send DNS + requests that are not "replayable" transactions". This specification + supports sending a NOTIFY in 0-RTT data because although a NOTIFY + technically changes the state of the receiving server, the effect of + replaying NOTIFYs has negligible impact in practice. + + NOTIFY messages prompt a secondary to either send an SOA query or an + XFR request to the primary on the basis that a newer version of the + zone is available. It has long been recognized that NOTIFYs can be + forged and, in theory, used to cause a secondary to send repeated + unnecessary requests to the primary. For this reason, most + implementations have some form of throttling of the SOA/XFR queries + triggered by the receipt of one or more NOTIFYs. + + [RFC9103] describes the privacy risks associated with both NOTIFY and + SOA queries and does not include addressing those risks within the + scope of encrypting zone transfers. Given this, the privacy benefit + of using DoQ for NOTIFY is not clear, but for the same reason, + sending NOTIFY as 0-RTT data has no privacy risk above that of + sending it using cleartext DNS. + +Acknowledgements + + This document liberally borrows text from the HTTP/3 specification + [HTTP/3] edited by Mike Bishop and from the DoT specification + [RFC7858] authored by Zi Hu, Liang Zhu, John Heidemann, Allison + Mankin, Duane Wessels, and Paul Hoffman. + + The privacy issue with 0-RTT data and session resumption was analyzed + by Daniel Kahn Gillmor (DKG) in a message to the IETF DPRIVE Working + Group [DNS0RTT]. + + Thanks to Tony Finch for an extensive review of the initial draft + version of this document, and to Robert Evans for the discussion of + 0-RTT privacy issues. Early reviews by Paul Hoffman and Martin + Thomson and interoperability tests conducted by Stephane Bortzmeyer + helped improve the definition of the protocol. + + Thanks also to Martin Thomson and Martin Duke for their later reviews + focusing on the low-level QUIC details, which helped clarify several + aspects of DoQ. Thanks to Andrey Meshkov, Loganaden Velvindron, + Lucas Pardue, Matt Joras, Mirja Kuelewind, Brian Trammell, and + Phillip Hallam-Baker for their reviews and contributions. + +Authors' Addresses + + Christian Huitema + Private Octopus Inc. + 427 Golfcourse Rd + Friday Harbor, WA 98250 + United States of America + Email: huitema@huitema.net + + + Sara Dickinson + Sinodun IT + Oxford Science Park + Oxford + OX4 4GA + United Kingdom + Email: sara@sinodun.com + + + Allison Mankin + Salesforce + Email: allison.mankin@gmail.com diff --git a/eval/corpora/rfc/RFC 9297 - HTTP Datagrams and the Capsule Protocol.txt b/eval/corpora/rfc/RFC 9297 - HTTP Datagrams and the Capsule Protocol.txt new file mode 100644 index 00000000..27349627 --- /dev/null +++ b/eval/corpora/rfc/RFC 9297 - HTTP Datagrams and the Capsule Protocol.txt @@ -0,0 +1,717 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) D. Schinazi +Request for Comments: 9297 Google LLC +Category: Standards Track L. Pardue +ISSN: 2070-1721 Cloudflare + August 2022 + + + HTTP Datagrams and the Capsule Protocol + +Abstract + + This document describes HTTP Datagrams, a convention for conveying + multiplexed, potentially unreliable datagrams inside an HTTP + connection. + + In HTTP/3, HTTP Datagrams can be sent unreliably using the QUIC + DATAGRAM extension. When the QUIC DATAGRAM frame is unavailable or + undesirable, HTTP Datagrams can be sent using the Capsule Protocol, + which is a more general convention for conveying data in HTTP + connections. + + HTTP Datagrams and the Capsule Protocol are intended for use by HTTP + extensions, not applications. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9297. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 1.1. Conventions and Definitions + 2. HTTP Datagrams + 2.1. HTTP/3 Datagrams + 2.1.1. The SETTINGS_H3_DATAGRAM HTTP/3 Setting + 2.2. HTTP Datagrams Using Capsules + 3. Capsules + 3.1. HTTP Data Streams + 3.2. The Capsule Protocol + 3.3. Error Handling + 3.4. The Capsule-Protocol Header Field + 3.5. The DATAGRAM Capsule + 4. Security Considerations + 5. IANA Considerations + 5.1. HTTP/3 Setting + 5.2. HTTP/3 Error Code + 5.3. HTTP Header Field Name + 5.4. Capsule Types + 6. References + 6.1. Normative References + 6.2. Informative References + Acknowledgments + Authors' Addresses + +1. Introduction + + HTTP extensions (as defined in Section 16 of [HTTP]) sometimes need + to access underlying transport protocol features such as unreliable + delivery (as offered by [QUIC-DGRAM]) to enable desirable features. + For example, this could allow for the introduction of an unreliable + version of the CONNECT method and the addition of unreliable delivery + to WebSockets [WEBSOCKET]. + + In Section 2, this document describes HTTP Datagrams, a convention + for conveying bidirectional and potentially unreliable datagrams + inside an HTTP connection, with multiplexing when possible. While + HTTP Datagrams are associated with HTTP requests, they are not a part + of message content. Instead, they are intended for use by HTTP + extensions (such as the CONNECT method) and are compatible with all + versions of HTTP. + + When HTTP is running over a transport protocol that supports + unreliable delivery (such as when the QUIC DATAGRAM extension + [QUIC-DGRAM] is available to HTTP/3 [HTTP/3]), HTTP Datagrams can use + that capability. + + In Section 3, this document describes the HTTP Capsule Protocol, + which allows the conveyance of HTTP Datagrams using reliable + delivery. This addresses HTTP/3 cases where use of the QUIC DATAGRAM + frame is unavailable or undesirable or where the transport protocol + only provides reliable delivery, such as with HTTP/1.1 [HTTP/1.1] or + HTTP/2 [HTTP/2] over TCP [TCP]. + +1.1. Conventions and Definitions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + This document uses terminology from [QUIC]. + + Where this document defines protocol types, the definition format + uses the notation from Section 1.3 of [QUIC]. Where fields within + types are integers, they are encoded using the variable-length + integer encoding from Section 16 of [QUIC]. Integer values do not + need to be encoded on the minimum number of bytes necessary. + + In this document, the term "intermediary" refers to an HTTP + intermediary as defined in Section 3.7 of [HTTP]. + +2. HTTP Datagrams + + HTTP Datagrams are a convention for conveying bidirectional and + potentially unreliable datagrams inside an HTTP connection with + multiplexing when possible. All HTTP Datagrams are associated with + an HTTP request. + + When HTTP Datagrams are conveyed on an HTTP/3 connection, the QUIC + DATAGRAM frame can be used to provide demultiplexing and unreliable + delivery; see Section 2.1. Negotiating the use of QUIC DATAGRAM + frames for HTTP Datagrams is achieved via the exchange of HTTP/3 + settings; see Section 2.1.1. + + When running over HTTP/2, demultiplexing is provided by the HTTP/2 + framing layer, but unreliable delivery is unavailable. HTTP + Datagrams are negotiated and conveyed using the Capsule Protocol; see + Section 3.5. + + When running over HTTP/1.x, requests are strictly serialized in the + connection; therefore, demultiplexing is not available. Unreliable + delivery is likewise not available. HTTP Datagrams are negotiated + and conveyed using the Capsule Protocol; see Section 3.5. + + HTTP Datagrams MUST only be sent with an association to an HTTP + request that explicitly supports them. For example, existing HTTP + methods GET and POST do not define semantics for associated HTTP + Datagrams; therefore, HTTP Datagrams associated with GET or POST + request streams cannot be sent. + + If an HTTP Datagram is received and it is associated with a request + that has no known semantics for HTTP Datagrams, the receiver MUST + terminate the request. If HTTP/3 is in use, the request stream MUST + be aborted with H3_DATAGRAM_ERROR (0x33). HTTP extensions MAY + override these requirements by defining a negotiation mechanism and + semantics for HTTP Datagrams. + +2.1. HTTP/3 Datagrams + + When used with HTTP/3, the Datagram Data field of QUIC DATAGRAM + frames uses the following format: + + HTTP/3 Datagram { + Quarter Stream ID (i), + HTTP Datagram Payload (..), + } + + Figure 1: HTTP/3 Datagram Format + + Quarter Stream ID: A variable-length integer that contains the value + of the client-initiated bidirectional stream that this datagram is + associated with divided by four (the division by four stems from + the fact that HTTP requests are sent on client-initiated + bidirectional streams, which have stream IDs that are divisible by + four). The largest legal QUIC stream ID value is 2^62-1, so the + largest legal value of the Quarter Stream ID field is 2^60-1. + Receipt of an HTTP/3 Datagram that includes a larger value MUST be + treated as an HTTP/3 connection error of type H3_DATAGRAM_ERROR + (0x33). + + HTTP Datagram Payload: The payload of the datagram, whose semantics + are defined by the extension that is using HTTP Datagrams. Note + that this field can be empty. + + Receipt of a QUIC DATAGRAM frame whose payload is too short to allow + parsing the Quarter Stream ID field MUST be treated as an HTTP/3 + connection error of type H3_DATAGRAM_ERROR (0x33). + + HTTP/3 Datagrams MUST NOT be sent unless the corresponding stream's + send side is open. If a datagram is received after the corresponding + stream's receive side is closed, the received datagrams MUST be + silently dropped. + + If an HTTP/3 Datagram is received and its Quarter Stream ID field + maps to a stream that has not yet been created, the receiver SHALL + either drop that datagram silently or buffer it temporarily (on the + order of a round trip) while awaiting the creation of the + corresponding stream. + + If an HTTP/3 Datagram is received and its Quarter Stream ID field + maps to a stream that cannot be created due to client-initiated + bidirectional stream limits, it SHOULD be treated as an HTTP/3 + connection error of type H3_ID_ERROR. Generating an error is not + mandatory because the QUIC stream limit might be unknown to the + HTTP/3 layer. + + Prioritization of HTTP/3 Datagrams is not defined in this document. + Future extensions MAY define how to prioritize datagrams and MAY + define signaling to allow communicating prioritization preferences. + +2.1.1. The SETTINGS_H3_DATAGRAM HTTP/3 Setting + + An endpoint can indicate to its peer that it is willing to receive + HTTP/3 Datagrams by sending the SETTINGS_H3_DATAGRAM (0x33) setting + with a value of 1. + + The value of the SETTINGS_H3_DATAGRAM setting MUST be either 0 or 1. + A value of 0 indicates that the implementation is not willing to + receive HTTP Datagrams. If the SETTINGS_H3_DATAGRAM setting is + received with a value that is neither 0 nor 1, the receiver MUST + terminate the connection with error H3_SETTINGS_ERROR. + + QUIC DATAGRAM frames MUST NOT be sent until the SETTINGS_H3_DATAGRAM + setting has been both sent and received with a value of 1. + + When clients use 0-RTT, they MAY store the value of the server's + SETTINGS_H3_DATAGRAM setting. Doing so allows the client to send + QUIC DATAGRAM frames in 0-RTT packets. When servers decide to accept + 0-RTT data, they MUST send a SETTINGS_H3_DATAGRAM setting greater + than or equal to the value they sent to the client in the connection + where they sent them the NewSessionTicket message. If a client + stores the value of the SETTINGS_H3_DATAGRAM setting with their 0-RTT + state, they MUST validate that the new value of the + SETTINGS_H3_DATAGRAM setting sent by the server in the handshake is + greater than or equal to the stored value; if not, the client MUST + terminate the connection with error H3_SETTINGS_ERROR. In all cases, + the maximum permitted value of the SETTINGS_H3_DATAGRAM setting + parameter is 1. + + It is RECOMMENDED that implementations that support receiving HTTP/3 + Datagrams always send the SETTINGS_H3_DATAGRAM setting with a value + of 1, even if the application does not intend to use HTTP/3 + Datagrams. This helps to avoid "sticking out"; see Section 4. + +2.2. HTTP Datagrams Using Capsules + + When HTTP/3 Datagrams are unavailable or undesirable, HTTP Datagrams + can be sent using the Capsule Protocol; see Section 3.5. + +3. Capsules + + One mechanism to extend HTTP is to introduce new HTTP upgrade tokens; + see Section 16.7 of [HTTP]. In HTTP/1.x, these tokens are used via + the Upgrade mechanism; see Section 7.8 of [HTTP]. In HTTP/2 and + HTTP/3, these tokens are used via the Extended CONNECT mechanism; see + [EXT-CONNECT2] and [EXT-CONNECT3]. + + This specification introduces the Capsule Protocol. The Capsule + Protocol is a sequence of type-length-value tuples that definitions + of new HTTP upgrade tokens can choose to use. It allows endpoints to + reliably communicate request-related information end-to-end on HTTP + request streams, even in the presence of HTTP intermediaries. The + Capsule Protocol can be used to exchange HTTP Datagrams, which is + necessary when HTTP is running over a transport that does not support + the QUIC DATAGRAM frame. The Capsule Protocol can also be used to + communicate reliable and bidirectional control messages associated + with a datagram-based protocol even when HTTP/3 Datagrams are in use. + +3.1. HTTP Data Streams + + This specification defines the "data stream" of an HTTP request as + the bidirectional stream of bytes that follows the header section of + the request message and the final response message that is either + successful (i.e., 2xx) or upgraded (i.e., 101). + + In HTTP/1.x, the data stream consists of all bytes on the connection + that follow the blank line that concludes either the request header + section or the final response header section. As a result, only the + last HTTP request on an HTTP/1.x connection can start the Capsule + Protocol. + + In HTTP/2 and HTTP/3, the data stream of a given HTTP request + consists of all bytes sent in DATA frames with the corresponding + stream ID. + + The concept of a data stream is particularly relevant for methods + such as CONNECT, where there is no HTTP message content after the + headers. + + Data streams can be prioritized using any means suited to stream or + request prioritization. For example, see Section 11 of [PRIORITY]. + + Data streams are subject to the flow control mechanisms of the + underlying layers; examples include HTTP/2 stream flow control, + HTTP/2 connection flow control, and TCP flow control. + +3.2. The Capsule Protocol + + Definitions of new HTTP upgrade tokens can state that their + associated request's data stream uses the Capsule Protocol. If they + do so, the contents of the associated request's data stream uses the + following format: + + Capsule Protocol { + Capsule (..) ..., + } + + Figure 2: Capsule Protocol Stream Format + + Capsule { + Capsule Type (i), + Capsule Length (i), + Capsule Value (..), + } + + Figure 3: Capsule Format + + Capsule Type: A variable-length integer indicating the type of the + capsule. An IANA registry is used to manage the assignment of + Capsule Types; see Section 5.4. + + Capsule Length: The length, in bytes, of the Capsule Value field, + which follows this field, encoded as a variable-length integer. + Note that this field can have a value of zero. + + Capsule Value: The payload of this Capsule. Its semantics are + determined by the value of the Capsule Type field. + + An intermediary can identify the use of the Capsule Protocol either + through the presence of the Capsule-Protocol header field + (Section 3.4) or by understanding the chosen HTTP Upgrade token. + + Because new protocols or extensions might define new Capsule Types, + intermediaries that wish to allow for future extensibility SHOULD + forward Capsules without modification unless the definition of the + Capsule Type in use specifies additional intermediary processing. + One such Capsule Type is the DATAGRAM Capsule; see Section 3.5. In + particular, intermediaries SHOULD forward Capsules with an unknown + Capsule Type without modification. + + Endpoints that receive a Capsule with an unknown Capsule Type MUST + silently drop that Capsule and skip over it to parse the next + Capsule. + + By virtue of the definition of the data stream: + + * The Capsule Protocol is not in use unless the response includes a + 2xx (Successful) or 101 (Switching Protocols) status code. + + * When the Capsule Protocol is in use, the associated HTTP request + and response do not carry HTTP content. A future extension MAY + define a new Capsule Type to carry HTTP content. + + The Capsule Protocol only applies to definitions of new HTTP upgrade + tokens; thus, in HTTP/2 and HTTP/3, it can only be used with the + CONNECT method. Therefore, once both endpoints agree to use the + Capsule Protocol, the frame usage requirements of the stream change + as specified in Section 8.5 of [HTTP/2] and Section 4.4 of [HTTP/3]. + + The Capsule Protocol MUST NOT be used with messages that contain + Content-Length, Content-Type, or Transfer-Encoding header fields. + Additionally, HTTP status codes 204 (No Content), 205 (Reset + Content), and 206 (Partial Content) MUST NOT be sent on responses + that use the Capsule Protocol. A receiver that observes a violation + of these requirements MUST treat the HTTP message as malformed. + + When processing Capsules, a receiver might be tempted to accumulate + the full length of the Capsule Value field in the data stream before + handling it. This approach SHOULD be avoided because it can consume + flow control in underlying layers, and that might lead to deadlocks + if the Capsule data exhausts the flow control window. + +3.3. Error Handling + + When a receiver encounters an error processing the Capsule Protocol, + the receiver MUST treat it as if it had received a malformed or + incomplete HTTP message. For HTTP/3, the handling of malformed + messages is described in Section 4.1.2 of [HTTP/3]. For HTTP/2, the + handling of malformed messages is described in Section 8.1.1 of + [HTTP/2]. For HTTP/1.x, the handling of incomplete messages is + described in Section 8 of [HTTP/1.1]. + + Each Capsule's payload MUST contain exactly the fields identified in + its description. A Capsule payload that contains additional bytes + after the identified fields or a Capsule payload that terminates + before the end of the identified fields MUST be treated as it if were + a malformed or incomplete message. In particular, redundant length + encodings MUST be verified to be self-consistent. + + If the receive side of a stream carrying Capsules is terminated + cleanly (for example, in HTTP/3 this is defined as receiving a QUIC + STREAM frame with the FIN bit set) and the last Capsule on the stream + was truncated, this MUST be treated as if it were a malformed or + incomplete message. + +3.4. The Capsule-Protocol Header Field + + The "Capsule-Protocol" header field is an Item Structured Field; see + Section 3.3 of [STRUCTURED-FIELDS]. Its value MUST be a Boolean; any + other value type MUST be handled as if the field were not present by + recipients (for example, if this field is included multiple times, + its type will become a List and the field will be ignored). This + document does not define any parameters for the Capsule-Protocol + header field value, but future documents might define parameters. + Receivers MUST ignore unknown parameters. + + Endpoints indicate that the Capsule Protocol is in use on a data + stream by sending a Capsule-Protocol header field with a true value. + A Capsule-Protocol header field with a false value has the same + semantics as when the header is not present. + + Intermediaries MAY use this header field to allow processing of HTTP + Datagrams for unknown HTTP upgrade tokens. Note that this is only + possible for HTTP Upgrade or Extended CONNECT. + + The Capsule-Protocol header field MUST NOT be used on HTTP responses + with a status code that is both different from 101 (Switching + Protocols) and outside the 2xx (Successful) range. + + When using the Capsule Protocol, HTTP endpoints SHOULD send the + Capsule-Protocol header field to simplify intermediary processing. + Definitions of new HTTP upgrade tokens that use the Capsule Protocol + MAY alter this recommendation. + +3.5. The DATAGRAM Capsule + + This document defines the DATAGRAM (0x00) Capsule Type. This Capsule + allows HTTP Datagrams to be sent on a stream using the Capsule + Protocol. This is particularly useful when HTTP is running over a + transport that does not support the QUIC DATAGRAM frame. + + Datagram Capsule { + Type (i) = 0x00, + Length (i), + HTTP Datagram Payload (..), + } + + Figure 4: DATAGRAM Capsule Format + + HTTP Datagram Payload: The payload of the datagram, whose semantics + are defined by the extension that is using HTTP Datagrams. Note + that this field can be empty. + + HTTP Datagrams sent using the DATAGRAM Capsule have the same + semantics as those sent in QUIC DATAGRAM frames. In particular, the + restrictions on when it is allowed to send an HTTP Datagram and how + to process them (from Section 2.1) also apply to HTTP Datagrams sent + and received using the DATAGRAM Capsule. + + An intermediary can re-encode HTTP Datagrams as it forwards them. In + other words, an intermediary MAY send a DATAGRAM Capsule to forward + an HTTP Datagram that was received in a QUIC DATAGRAM frame and vice + versa. Intermediaries MUST NOT perform this re-encoding unless they + have identified the use of the Capsule Protocol on the corresponding + request stream; see Section 3.2. + + Note that while DATAGRAM Capsules, which are sent on a stream, are + reliably delivered in order, intermediaries can re-encode DATAGRAM + Capsules into QUIC DATAGRAM frames when forwarding messages, which + could result in loss or reordering. + + If an intermediary receives an HTTP Datagram in a QUIC DATAGRAM frame + and is forwarding it on a connection that supports QUIC DATAGRAM + frames, the intermediary SHOULD NOT convert that HTTP Datagram to a + DATAGRAM Capsule. If the HTTP Datagram is too large to fit in a + DATAGRAM frame (for example, because the Path MTU (PMTU) of that QUIC + connection is too low or if the maximum UDP payload size advertised + on that connection is too low), the intermediary SHOULD drop the HTTP + Datagram instead of converting it to a DATAGRAM Capsule. This + preserves the end-to-end unreliability characteristic that methods + such as Datagram Packetization Layer PMTU Discovery (DPLPMTUD) depend + on [DPLPMTUD]. An intermediary that converts QUIC DATAGRAM frames to + DATAGRAM Capsules allows HTTP Datagrams to be arbitrarily large + without suffering any loss. This can misrepresent the true path + properties, defeating methods such as DPLPMTUD. + + While DATAGRAM Capsules can theoretically carry a payload of length + 2^62-1, most HTTP extensions that use HTTP Datagrams will have their + own limits on what datagram payload sizes are practical. + Implementations SHOULD take those limits into account when parsing + DATAGRAM Capsules. If an incoming DATAGRAM Capsule has a length that + is known to be so large as to not be usable, the implementation + SHOULD discard the Capsule without buffering its contents into + memory. + + Since QUIC DATAGRAM frames are required to fit within a QUIC packet, + implementations that re-encode DATAGRAM Capsules into QUIC DATAGRAM + frames might be tempted to accumulate the entire Capsule in the + stream before re-encoding it. This SHOULD be avoided, because it can + cause flow control problems; see Section 3.2. + + Note that it is possible for an HTTP extension to use HTTP Datagrams + without using the Capsule Protocol. For example, if an HTTP + extension that uses HTTP Datagrams is only defined over transports + that support QUIC DATAGRAM frames, it might not need a stream + encoding. Additionally, HTTP extensions can use HTTP Datagrams with + their own data stream protocol. However, new HTTP extensions that + wish to use HTTP Datagrams SHOULD use the Capsule Protocol, as + failing to do so will make it harder for the HTTP extension to + support versions of HTTP other than HTTP/3 and will prevent + interoperability with intermediaries that only support the Capsule + Protocol. + +4. Security Considerations + + Since transmitting HTTP Datagrams using QUIC DATAGRAM frames requires + sending the HTTP/3 SETTINGS_H3_DATAGRAM setting, it "sticks out". In + other words, probing clients can learn whether a server supports HTTP + Datagrams over QUIC DATAGRAM frames. As some servers might wish to + obfuscate the fact that they offer application services that use HTTP + Datagrams, it's best for all implementations that support this + feature to always send this setting; see Section 2.1.1. + + Since use of the Capsule Protocol is restricted to new HTTP upgrade + tokens, it is not directly accessible from Web Platform APIs (such as + those commonly accessed via JavaScript in web browsers). + + Definitions of new HTTP upgrade tokens that use the Capsule Protocol + need to include a security analysis that considers the impact of HTTP + Datagrams and Capsules in the context of their protocol. + +5. IANA Considerations + +5.1. HTTP/3 Setting + + IANA has registered the following entry in the "HTTP/3 Settings" + registry maintained at <https://www.iana.org/assignments/ + http3-parameters>: + + Value: 0x33 + Setting Name: SETTINGS_H3_DATAGRAM + Default: 0 + Status: permanent + Reference: RFC 9297 + Change Controller: IETF + Contact: HTTP_WG; HTTP working group; ietf-http-wg@w3.org + Notes: None + +5.2. HTTP/3 Error Code + + IANA has registered the following entry in the "HTTP/3 Error Codes" + registry maintained at <https://www.iana.org/assignments/ + http3-parameters>: + + Value: 0x33 + Name: H3_DATAGRAM_ERROR + Description: Datagram or Capsule Protocol parse error + Status: permanent + Reference: RFC 9297 + Change Controller: IETF + Contact: HTTP_WG; HTTP working group; ietf-http-wg@w3.org + Notes: None + +5.3. HTTP Header Field Name + + IANA has registered the following entry in the "Hypertext Transfer + Protocol (HTTP) Field Name Registry" maintained at + <https://www.iana.org/assignments/http-fields>: + + Field Name: Capsule-Protocol + Template: None + Status: permanent + Reference: RFC 9297 + Comments: None + +5.4. Capsule Types + + This document establishes a registry for HTTP Capsule Type codes. + The "HTTP Capsule Types" registry governs a 62-bit space and operates + under the QUIC registration policy documented in Section 22.1 of + [QUIC]. This new registry includes the common set of fields listed + in Section 22.1.1 of [QUIC]. In addition to those common fields, all + registrations in this registry MUST include a "Capsule Type" field + that contains a short name or label for the Capsule Type. + + Permanent registrations in this registry are assigned using the + Specification Required policy (Section 4.6 of [IANA-POLICY]), except + for values between 0x00 and 0x3f (in hexadecimal; inclusive), which + are assigned using Standards Action or IESG Approval as defined in + Sections 4.9 and 4.10 of [IANA-POLICY]. + + Capsule Types with a value of the form 0x29 * N + 0x17 for integer + values of N are reserved to exercise the requirement that unknown + Capsule Types be ignored. These Capsules have no semantics and can + carry arbitrary values. These values MUST NOT be assigned by IANA + and MUST NOT appear in the listing of assigned values. + + This registry initially contains the following entry: + + Value: 0x00 + Capsule Type: DATAGRAM + Status: permanent + Reference: RFC 9297 + Change Controller: IETF + Contact: MASQUE Working Group masque@ietf.org + (mailto:masque@ietf.org) + Notes: None + +6. References + +6.1. Normative References + + [HTTP] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP Semantics", STD 97, RFC 9110, + DOI 10.17487/RFC9110, June 2022, + <https://www.rfc-editor.org/info/rfc9110>. + + [HTTP/1.1] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP/1.1", STD 99, RFC 9112, DOI 10.17487/RFC9112, + June 2022, <https://www.rfc-editor.org/info/rfc9112>. + + [HTTP/2] Thomson, M., Ed. and C. Benfield, Ed., "HTTP/2", RFC 9113, + DOI 10.17487/RFC9113, June 2022, + <https://www.rfc-editor.org/info/rfc9113>. + + [HTTP/3] Bishop, M., Ed., "HTTP/3", RFC 9114, DOI 10.17487/RFC9114, + June 2022, <https://www.rfc-editor.org/info/rfc9114>. + + [IANA-POLICY] + Cotton, M., Leiba, B., and T. Narten, "Guidelines for + Writing an IANA Considerations Section in RFCs", BCP 26, + RFC 8126, DOI 10.17487/RFC8126, June 2017, + <https://www.rfc-editor.org/info/rfc8126>. + + [QUIC] Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [QUIC-DGRAM] + Pauly, T., Kinnear, E., and D. Schinazi, "An Unreliable + Datagram Extension to QUIC", RFC 9221, + DOI 10.17487/RFC9221, March 2022, + <https://www.rfc-editor.org/info/rfc9221>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + + [STRUCTURED-FIELDS] + Nottingham, M. and P-H. Kamp, "Structured Field Values for + HTTP", RFC 8941, DOI 10.17487/RFC8941, February 2021, + <https://www.rfc-editor.org/info/rfc8941>. + + [TCP] Eddy, W., Ed., "Transmission Control Protocol (TCP)", + STD 7, RFC 9293, DOI 10.17487/RFC9293, August 2022, + <https://www.rfc-editor.org/info/rfc9293>. + +6.2. Informative References + + [DPLPMTUD] Fairhurst, G., Jones, T., Tรผxen, M., Rรผngeler, I., and T. + Vรถlker, "Packetization Layer Path MTU Discovery for + Datagram Transports", RFC 8899, DOI 10.17487/RFC8899, + September 2020, <https://www.rfc-editor.org/info/rfc8899>. + + [EXT-CONNECT2] + McManus, P., "Bootstrapping WebSockets with HTTP/2", + RFC 8441, DOI 10.17487/RFC8441, September 2018, + <https://www.rfc-editor.org/info/rfc8441>. + + [EXT-CONNECT3] + Hamilton, R., "Bootstrapping WebSockets with HTTP/3", + RFC 9220, DOI 10.17487/RFC9220, June 2022, + <https://www.rfc-editor.org/info/rfc9220>. + + [PRIORITY] Oku, K. and L. Pardue, "Extensible Prioritization Scheme + for HTTP", RFC 9218, DOI 10.17487/RFC9218, June 2022, + <https://www.rfc-editor.org/info/rfc9218>. + + [WEBSOCKET] + Fette, I. and A. Melnikov, "The WebSocket Protocol", + RFC 6455, DOI 10.17487/RFC6455, December 2011, + <https://www.rfc-editor.org/info/rfc6455>. + +Acknowledgments + + Portions of this document were previously part of the QUIC DATAGRAM + frame definition itself; the authors would like to acknowledge the + authors of that document and the members of the IETF MASQUE working + group for their suggestions. Additionally, the authors would like to + thank Martin Thomson for suggesting the use of an HTTP/3 setting. + Furthermore, the authors would like to thank Ben Schwartz for + substantive input. The final design in this document came out of the + HTTP Datagrams Design Team, whose members were Alan Frindell, Alex + Chernyakhovsky, Ben Schwartz, Eric Rescorla, Marcus Ihlar, Martin + Thomson, Mike Bishop, Tommy Pauly, Victor Vasiliev, and the authors + of this document. The authors thank Mark Nottingham and Philipp + Tiesel for their helpful comments. + +Authors' Addresses + + David Schinazi + Google LLC + 1600 Amphitheatre Parkway + Mountain View, CA 94043 + United States of America + Email: dschinazi.ietf@gmail.com + + + Lucas Pardue + Cloudflare + Email: lucaspardue.24.7@gmail.com diff --git a/eval/corpora/rfc/RFC 9298 - Proxying UDP in HTTP.txt b/eval/corpora/rfc/RFC 9298 - Proxying UDP in HTTP.txt new file mode 100644 index 00000000..c383a434 --- /dev/null +++ b/eval/corpora/rfc/RFC 9298 - Proxying UDP in HTTP.txt @@ -0,0 +1,819 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) D. Schinazi +Request for Comments: 9298 Google LLC +Category: Standards Track August 2022 +ISSN: 2070-1721 + + + Proxying UDP in HTTP + +Abstract + + This document describes how to proxy UDP in HTTP, similar to how the + HTTP CONNECT method allows proxying TCP in HTTP. More specifically, + this document defines a protocol that allows an HTTP client to create + a tunnel for UDP communications through an HTTP server that acts as a + proxy. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9298. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 1.1. Conventions and Definitions + 2. Client Configuration + 3. Tunneling UDP over HTTP + 3.1. UDP Proxy Handling + 3.2. HTTP/1.1 Request + 3.3. HTTP/1.1 Response + 3.4. HTTP/2 and HTTP/3 Requests + 3.5. HTTP/2 and HTTP/3 Responses + 4. Context Identifiers + 5. HTTP Datagram Payload Format + 6. Performance Considerations + 6.1. MTU Considerations + 6.2. Tunneling of ECN Marks + 7. Security Considerations + 8. IANA Considerations + 8.1. HTTP Upgrade Token + 8.2. Well-Known URI + 9. References + 9.1. Normative References + 9.2. Informative References + Acknowledgments + Author's Address + +1. Introduction + + While HTTP provides the CONNECT method (see Section 9.3.6 of [HTTP]) + for creating a TCP [TCP] tunnel to a proxy, it lacked a method for + doing so for UDP [UDP] traffic prior to this specification. + + This document describes a protocol for tunneling UDP to a server + acting as a UDP-specific proxy over HTTP. UDP tunnels are commonly + used to create an end-to-end virtual connection, which can then be + secured using QUIC [QUIC] or another protocol running over UDP. + Unlike the HTTP CONNECT method, the UDP proxy itself is identified + with an absolute URL containing the traffic's destination. Clients + generate those URLs using a URI Template [TEMPLATE], as described in + Section 2. + + This protocol supports all existing versions of HTTP by using HTTP + Datagrams [HTTP-DGRAM]. When using HTTP/2 [HTTP/2] or HTTP/3 + [HTTP/3], it uses HTTP Extended CONNECT as described in + [EXT-CONNECT2] and [EXT-CONNECT3]. When using HTTP/1.x [HTTP/1.1], + it uses HTTP Upgrade as defined in Section 7.8 of [HTTP]. + +1.1. Conventions and Definitions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + In this document, we use the term "UDP proxy" to refer to the HTTP + server that acts upon the client's UDP tunneling request to open a + UDP socket to a target server and that generates the response to this + request. If there are HTTP intermediaries (as defined in Section 3.7 + of [HTTP]) between the client and the UDP proxy, those are referred + to as "intermediaries" in this document. + + Note that, when the HTTP version in use does not support multiplexing + streams (such as HTTP/1.1), any reference to "stream" in this + document represents the entire connection. + +2. Client Configuration + + HTTP clients are configured to use a UDP proxy with a URI Template + [TEMPLATE] that has the variables "target_host" and "target_port". + Examples are shown below: + + https://example.org/.well-known/masque/udp/{target_host}/{target_port}/ + https://proxy.example.org:4443/masque?h={target_host}&p={target_port} + https://proxy.example.org:4443/masque{?target_host,target_port} + + Figure 1: URI Template Examples + + The following requirements apply to the URI Template: + + * The URI Template MUST be a level 3 template or lower. + + * The URI Template MUST be in absolute form and MUST include non- + empty scheme, authority, and path components. + + * The path component of the URI Template MUST start with a slash + ("/"). + + * All template variables MUST be within the path or query components + of the URI. + + * The URI Template MUST contain the two variables "target_host" and + "target_port" and MAY contain other variables. + + * The URI Template MUST NOT contain any non-ASCII Unicode characters + and MUST only contain ASCII characters in the range 0x21-0x7E + inclusive (note that percent-encoding is allowed; see Section 2.1 + of [URI]). + + * The URI Template MUST NOT use Reserved Expansion ("+" operator), + Fragment Expansion ("#" operator), Label Expansion with Dot- + Prefix, Path Segment Expansion with Slash-Prefix, nor Path-Style + Parameter Expansion with Semicolon-Prefix. + + Clients SHOULD validate the requirements above; however, clients MAY + use a general-purpose URI Template implementation that lacks this + specific validation. If a client detects that any of the + requirements above are not met by a URI Template, the client MUST + reject its configuration and abort the request without sending it to + the UDP proxy. + + The original HTTP CONNECT method allowed for the conveyance of the + target host and port, but not the scheme, proxy authority, path, or + query. Thus, clients with proxy configuration interfaces that only + allow the user to configure the proxy host and the proxy port exist. + Client implementations of this specification that are constrained by + such limitations MAY attempt to access UDP proxying capabilities + using the default template, which is defined as + "https://$PROXY_HOST:$PROXY_PORT/.well-known/masque/ + udp/{target_host}/{target_port}/", where $PROXY_HOST and $PROXY_PORT + are the configured host and port of the UDP proxy, respectively. UDP + proxy deployments SHOULD offer service at this location if they need + to interoperate with such clients. + +3. Tunneling UDP over HTTP + + To allow negotiation of a tunnel for UDP over HTTP, this document + defines the "connect-udp" HTTP upgrade token. The resulting UDP + tunnels use the Capsule Protocol (see Section 3.2 of [HTTP-DGRAM]) + with HTTP Datagrams in the format defined in Section 5. + + To initiate a UDP tunnel associated with a single HTTP stream, a + client issues a request containing the "connect-udp" upgrade token. + The target of the tunnel is indicated by the client to the UDP proxy + via the "target_host" and "target_port" variables of the URI + Template; see Section 2. + + "target_host" supports using DNS names, IPv6 literals and IPv4 + literals. Note that IPv6 scoped addressing zone identifiers are not + supported. Using the terms IPv6address, IPv4address, reg-name, and + port from [URI], the "target_host" and "target_port" variables MUST + adhere to the format in Figure 2, using notation from [ABNF]. + Additionally: + + * both the "target_host" and "target_port" variables MUST NOT be + empty. + + * if "target_host" contains an IPv6 literal, the colons (":") MUST + be percent-encoded. For example, if the target host is + "2001:db8::42", it will be encoded in the URI as + "2001%3Adb8%3A%3A42". + + * "target_port" MUST represent an integer between 1 and 65535 + inclusive. + + target_host = IPv6address / IPv4address / reg-name + target_port = port + + Figure 2: URI Template Variable Format + + When sending its UDP proxying request, the client SHALL perform URI + Template expansion to determine the path and query of its request. + + If the request is successful, the UDP proxy commits to converting + received HTTP Datagrams into UDP packets, and vice versa, until the + tunnel is closed. + + By virtue of the definition of the Capsule Protocol (see Section 3.2 + of [HTTP-DGRAM]), UDP proxying requests do not carry any message + content. Similarly, successful UDP proxying responses also do not + carry any message content. + +3.1. UDP Proxy Handling + + Upon receiving a UDP proxying request: + + * if the recipient is configured to use another HTTP proxy, it will + act as an intermediary by forwarding the request to another HTTP + server. Note that such intermediaries may need to re-encode the + request if they forward it using a version of HTTP that is + different from the one used to receive it, as the request encoding + differs by version (see below). + + * otherwise, the recipient will act as a UDP proxy. It extracts the + "target_host" and "target_port" variables from the URI it has + reconstructed from the request headers, decodes their percent- + encoding, and establishes a tunnel by directly opening a UDP + socket to the requested target. + + Unlike TCP, UDP is connectionless. The UDP proxy that opens the UDP + socket has no way of knowing whether the destination is reachable. + Therefore, it needs to respond to the request without waiting for a + packet from the target. However, if the "target_host" is a DNS name, + the UDP proxy MUST perform DNS resolution before replying to the HTTP + request. If errors occur during this process, the UDP proxy MUST + reject the request and SHOULD send details using an appropriate + Proxy-Status header field [PROXY-STATUS]. For example, if DNS + resolution returns an error, the proxy can use the dns_error Proxy + Error Type from Section 2.3.2 of [PROXY-STATUS]. + + UDP proxies can use connected UDP sockets if their operating system + supports them, as that allows the UDP proxy to rely on the kernel to + only send it UDP packets that match the correct 5-tuple. If the UDP + proxy uses a non-connected socket, it MUST validate the IP source + address and UDP source port on received packets to ensure they match + the client's request. Packets that do not match MUST be discarded by + the UDP proxy. + + The lifetime of the socket is tied to the request stream. The UDP + proxy MUST keep the socket open while the request stream is open. If + a UDP proxy is notified by its operating system that its socket is no + longer usable, it MUST close the request stream. For example, this + can happen when an ICMP Destination Unreachable message is received; + see Section 3.1 of [ICMP6]. UDP proxies MAY choose to close sockets + due to a period of inactivity, but they MUST close the request stream + when closing the socket. UDP proxies that close sockets after a + period of inactivity SHOULD NOT use a period lower than two minutes; + see Section 4.3 of [BEHAVE]. + + A successful response (as defined in Sections 3.3 and 3.5) indicates + that the UDP proxy has opened a socket to the requested target and is + willing to proxy UDP payloads. Any response other than a successful + response indicates that the request has failed; thus, the client MUST + abort the request. + + UDP proxies MUST NOT introduce fragmentation at the IP layer when + forwarding HTTP Datagrams onto a UDP socket; overly large datagrams + are silently dropped. In IPv4, the Don't Fragment (DF) bit MUST be + set, if possible, to prevent fragmentation on the path. Future + extensions MAY remove these requirements. + + Implementers of UDP proxies will benefit from reading the guidance in + [UDP-USAGE]. + +3.2. HTTP/1.1 Request + + When using HTTP/1.1 [HTTP/1.1], a UDP proxying request will meet the + following requirements: + + * the method SHALL be "GET". + + * the request SHALL include a single Host header field containing + the origin of the UDP proxy. + + * the request SHALL include a Connection header field with value + "Upgrade" (note that this requirement is case-insensitive as per + Section 7.6.1 of [HTTP]). + + * the request SHALL include an Upgrade header field with value + "connect-udp". + + A UDP proxying request that does not conform to these restrictions is + malformed. The recipient of such a malformed request MUST respond + with an error and SHOULD use the 400 (Bad Request) status code. + + For example, if the client is configured with URI Template + "https://example.org/.well-known/masque/ + udp/{target_host}/{target_port}/" and wishes to open a UDP proxying + tunnel to target 192.0.2.6:443, it could send the following request: + + GET https://example.org/.well-known/masque/udp/192.0.2.6/443/ HTTP/1.1 + Host: example.org + Connection: Upgrade + Upgrade: connect-udp + Capsule-Protocol: ?1 + + Figure 3: Example HTTP/1.1 Request + + In HTTP/1.1, this protocol uses the GET method to mimic the design of + the WebSocket Protocol [WEBSOCKET]. + +3.3. HTTP/1.1 Response + + The UDP proxy SHALL indicate a successful response by replying with + the following requirements: + + * the HTTP status code on the response SHALL be 101 (Switching + Protocols). + + * the response SHALL include a Connection header field with value + "Upgrade" (note that this requirement is case-insensitive as per + Section 7.6.1 of [HTTP]). + + * the response SHALL include a single Upgrade header field with + value "connect-udp". + + * the response SHALL meet the requirements of HTTP responses that + start the Capsule Protocol; see Section 3.2 of [HTTP-DGRAM]. + + If any of these requirements are not met, the client MUST treat this + proxying attempt as failed and abort the connection. + + For example, the UDP proxy could respond with: + + HTTP/1.1 101 Switching Protocols + Connection: Upgrade + Upgrade: connect-udp + Capsule-Protocol: ?1 + + Figure 4: Example HTTP/1.1 Response + +3.4. HTTP/2 and HTTP/3 Requests + + When using HTTP/2 [HTTP/2] or HTTP/3 [HTTP/3], UDP proxying requests + use HTTP Extended CONNECT. This requires that servers send an HTTP + Setting as specified in [EXT-CONNECT2] and [EXT-CONNECT3] and that + requests use HTTP pseudo-header fields with the following + requirements: + + * The :method pseudo-header field SHALL be "CONNECT". + + * The :protocol pseudo-header field SHALL be "connect-udp". + + * The :authority pseudo-header field SHALL contain the authority of + the UDP proxy. + + * The :path and :scheme pseudo-header fields SHALL NOT be empty. + Their values SHALL contain the scheme and path from the URI + Template after the URI Template expansion process has been + completed. + + A UDP proxying request that does not conform to these restrictions is + malformed (see Section 8.1.1 of [HTTP/2] and Section 4.1.2 of + [HTTP/3]). + + For example, if the client is configured with URI Template + "https://example.org/.well-known/masque/ + udp/{target_host}/{target_port}/" and wishes to open a UDP proxying + tunnel to target 192.0.2.6:443, it could send the following request: + + HEADERS + :method = CONNECT + :protocol = connect-udp + :scheme = https + :path = /.well-known/masque/udp/192.0.2.6/443/ + :authority = example.org + capsule-protocol = ?1 + + Figure 5: Example HTTP/2 Request + +3.5. HTTP/2 and HTTP/3 Responses + + The UDP proxy SHALL indicate a successful response by replying with + the following requirements: + + * the HTTP status code on the response SHALL be in the 2xx + (Successful) range. + + * the response SHALL meet the requirements of HTTP responses that + start the Capsule Protocol; see Section 3.2 of [HTTP-DGRAM]. + + If any of these requirements are not met, the client MUST treat this + proxying attempt as failed and abort the request. + + For example, the UDP proxy could respond with: + + HEADERS + :status = 200 + capsule-protocol = ?1 + + Figure 6: Example HTTP/2 Response + +4. Context Identifiers + + The mechanism for proxying UDP in HTTP defined in this document + allows future extensions to exchange HTTP Datagrams that carry + different semantics from UDP payloads. Some of these extensions can + augment UDP payloads with additional data, while others can exchange + data that is completely separate from UDP payloads. In order to + accomplish this, all HTTP Datagrams associated with UDP Proxying + request streams start with a Context ID field; see Section 5. + + Context IDs are 62-bit integers (0 to 2^62-1). Context IDs are + encoded as variable-length integers; see Section 16 of [QUIC]. The + Context ID value of 0 is reserved for UDP payloads, while non-zero + values are dynamically allocated. Non-zero even-numbered Context IDs + are client-allocated, and odd-numbered Context IDs are proxy- + allocated. The Context ID namespace is tied to a given HTTP request; + it is possible for a Context ID with the same numeric value to be + simultaneously allocated in distinct requests, potentially with + different semantics. Context IDs MUST NOT be re-allocated within a + given HTTP namespace but MAY be allocated in any order. The Context + ID allocation restrictions to the use of even-numbered and odd- + numbered Context IDs exist in order to avoid the need for + synchronization between endpoints. However, once a Context ID has + been allocated, those restrictions do not apply to the use of the + Context ID; it can be used by any client or UDP proxy, independent of + which endpoint initially allocated it. + + Registration is the action by which an endpoint informs its peer of + the semantics and format of a given Context ID. This document does + not define how registration occurs. Future extensions MAY use HTTP + header fields or capsules to register Context IDs. Depending on the + method being used, it is possible for datagrams to be received with + Context IDs that have not yet been registered. For instance, this + can be due to reordering of the packet containing the datagram and + the packet containing the registration message during transmission. + +5. HTTP Datagram Payload Format + + When HTTP Datagrams (see Section 2 of [HTTP-DGRAM]) are associated + with UDP Proxying request streams, the HTTP Datagram Payload field + has the format defined in Figure 7, using notation from Section 1.3 + of [QUIC]. Note that when HTTP Datagrams are encoded using QUIC + DATAGRAM frames [QUIC-DGRAM], the Context ID field defined below + directly follows the Quarter Stream ID field, which is at the start + of the QUIC DATAGRAM frame payload; see Section 2.1 of [HTTP-DGRAM]. + + UDP Proxying HTTP Datagram Payload { + Context ID (i), + UDP Proxying Payload (..), + } + + Figure 7: UDP Proxying HTTP Datagram Format + + Context ID: A variable-length integer (see Section 16 of [QUIC]) + that contains the value of the Context ID. If an HTTP/3 Datagram + that carries an unknown Context ID is received, the receiver SHALL + either drop that datagram silently or buffer it temporarily (on + the order of a round trip) while awaiting the registration of the + corresponding Context ID. + UDP Proxying Payload: The payload of the datagram, whose semantics + depend on the value of the previous field. Note that this field + can be empty. + + UDP packets are encoded using HTTP Datagrams with the Context ID + field set to zero. When the Context ID field is set to zero, the UDP + Proxying Payload field contains the unmodified payload of a UDP + packet (referred to as data octets in [UDP]). + + By virtue of the definition of the UDP header [UDP], it is not + possible to encode UDP payloads longer than 65527 bytes. Therefore, + endpoints MUST NOT send HTTP Datagrams with a UDP Proxying Payload + field longer than 65527 using Context ID zero. An endpoint that + receives an HTTP Datagram using Context ID zero whose UDP Proxying + Payload field is longer than 65527 MUST abort the corresponding + stream. If a UDP proxy knows it can only send out UDP packets of a + certain length due to its underlying link MTU, it has no choice but + to discard incoming HTTP Datagrams using Context ID zero whose UDP + Proxying Payload field is longer than that limit. If the discarded + HTTP Datagram was transported by a DATAGRAM capsule, the receiver + SHOULD discard that capsule without buffering the capsule contents. + + If a UDP proxy receives an HTTP Datagram before it has received the + corresponding request, it SHALL either drop that HTTP Datagram + silently or buffer it temporarily (on the order of a round trip) + while awaiting the corresponding request. + + Note that buffering datagrams (either because the request was not yet + received or because the Context ID is not yet known) consumes + resources. Receivers that buffer datagrams SHOULD apply buffering + limits in order to reduce the risk of resource exhaustion occurring. + For example, receivers can limit the total number of buffered + datagrams or the cumulative size of buffered datagrams on a per- + stream, per-context, or per-connection basis. + + A client MAY optimistically start sending UDP packets in HTTP + Datagrams before receiving the response to its UDP proxying request. + However, implementers should note that such proxied packets may not + be processed by the UDP proxy if it responds to the request with a + failure or if the proxied packets are received by the UDP proxy + before the request and the UDP proxy chooses to not buffer them. + +6. Performance Considerations + + Bursty traffic can often lead to temporally correlated packet losses; + in turn, this can lead to suboptimal responses from congestion + controllers in protocols running over UDP. To avoid this, UDP + proxies SHOULD strive to avoid increasing burstiness of UDP traffic; + they SHOULD NOT queue packets in order to increase batching. + + When the protocol running over UDP that is being proxied uses + congestion control (e.g., [QUIC]), the proxied traffic will incur at + least two nested congestion controllers. The underlying HTTP + connection MUST NOT disable congestion control unless it has an out- + of-band way of knowing with absolute certainty that the inner traffic + is congestion-controlled. + + If a client or UDP proxy with a connection containing a UDP Proxying + request stream disables congestion control, it MUST NOT signal + Explicit Congestion Notification (ECN) [ECN] support on that + connection. That is, it MUST mark all IP headers with the Not-ECT + codepoint. It MAY continue to report ECN feedback via QUIC ACK_ECN + frames or the TCP ECE bit, as the peer may not have disabled + congestion control. + + When the protocol running over UDP that is being proxied uses loss + recovery (e.g., [QUIC]), and the underlying HTTP connection runs over + TCP, the proxied traffic will incur at least two nested loss recovery + mechanisms. This can reduce performance as both can sometimes + independently retransmit the same data. To avoid this, UDP proxying + SHOULD be performed over HTTP/3 to allow leveraging the QUIC DATAGRAM + frame. + +6.1. MTU Considerations + + When using HTTP/3 with the QUIC Datagram extension [QUIC-DGRAM], UDP + payloads are transmitted in QUIC DATAGRAM frames. Since those cannot + be fragmented, they can only carry payloads up to a given length + determined by the QUIC connection configuration and the Path MTU + (PMTU). If a UDP proxy is using QUIC DATAGRAM frames and it receives + a UDP payload from the target that will not fit inside a QUIC + DATAGRAM frame, the UDP proxy SHOULD NOT send the UDP payload in a + DATAGRAM capsule, as that defeats the end-to-end unreliability + characteristic that methods such as Datagram Packetization Layer PMTU + Discovery (DPLPMTUD) depend on [DPLPMTUD]. In this scenario, the UDP + proxy SHOULD drop the UDP payload and send an ICMP Packet Too Big + message to the target; see Section 3.2 of [ICMP6]. + +6.2. Tunneling of ECN Marks + + UDP proxying does not create an IP-in-IP tunnel, so the guidance in + [ECN-TUNNEL] about transferring ECN marks between inner and outer IP + headers does not apply. There is no inner IP header in UDP proxying + tunnels. + + In this specification, note that UDP proxying clients do not have the + ability to control the ECN codepoints on UDP packets the UDP proxy + sends to the target, nor can UDP proxies communicate the markings of + each UDP packet from target to UDP proxy. + + A UDP proxy MUST ignore ECN bits in the IP header of UDP packets + received from the target, and it MUST set the ECN bits to Not-ECT on + UDP packets it sends to the target. These do not relate to the ECN + markings of packets sent between client and UDP proxy in any way. + +7. Security Considerations + + There are significant risks in allowing arbitrary clients to + establish a tunnel to arbitrary targets, as that could allow bad + actors to send traffic and have it attributed to the UDP proxy. HTTP + servers that support UDP proxying ought to restrict its use to + authenticated users. + + There exist software and network deployments that perform access + control checks based on the source IP address of incoming requests. + For example, some software allows unauthenticated configuration + changes if they originated from 127.0.0.1. Such software could be + running on the same host as the UDP proxy or in the same broadcast + domain. Proxied UDP traffic would then be received with a source IP + address belonging to the UDP proxy. If this source address is used + for access control, UDP proxying clients could use the UDP proxy to + escalate their access privileges beyond those they might otherwise + have. This could lead to unauthorized access by UDP proxying clients + unless the UDP proxy disallows UDP proxying requests to vulnerable + targets, such as the UDP proxy's own addresses and localhost, link- + local, multicast, and broadcast addresses. UDP proxies can use the + destination_ip_prohibited Proxy Error Type from Section 2.3.5 of + [PROXY-STATUS] when rejecting such requests. + + UDP proxies share many similarities with TCP CONNECT proxies when + considering them as infrastructure for abuse to enable denial-of- + service (DoS) attacks. Both can obfuscate the attacker's source + address from the attack target. In the case of a stateless + volumetric attack (e.g., a TCP SYN flood or a UDP flood), both types + of proxies pass the traffic to the target host. With stateful + volumetric attacks (e.g., HTTP flooding) being sent over a TCP + CONNECT proxy, the proxy will only send data if the target has + indicated its willingness to accept data by responding with a TCP + SYN-ACK. Once the path to the target is flooded, the TCP CONNECT + proxy will no longer receive replies from the target and will stop + sending data. Since UDP does not establish shared state between the + UDP proxy and the target, the UDP proxy could continue sending data + to the target in such a situation. While a UDP proxy could + potentially limit the number of UDP packets it is willing to forward + until it has observed a response from the target, that provides + limited protection against DoS attacks when attacks target open UDP + ports where the protocol running over UDP would respond and that + would be interpreted as willingness to accept UDP by the UDP proxy. + Such a packet limit could also cause issues for valid traffic. + + The security considerations described in Section 4 of [HTTP-DGRAM] + also apply here. Since it is possible to tunnel IP packets over UDP, + the guidance in [TUNNEL-SECURITY] can apply. + +8. IANA Considerations + +8.1. HTTP Upgrade Token + + IANA has registered "connect-udp" in the "HTTP Upgrade Tokens" + registry maintained at <https://www.iana.org/assignments/http- + upgrade-tokens>. + + Value: connect-udp + Description: Proxying of UDP Payloads + Expected Version Tokens: None + Reference: RFC 9298 + +8.2. Well-Known URI + + IANA has registered "masque" in the "Well-Known URIs" registry + maintained at <https://www.iana.org/assignments/well-known-uris>. + + URI Suffix: masque + Change Controller: IETF + Reference: RFC 9298 + Status: permanent + Related Information: Includes all resources identified with the path + prefix "/.well-known/masque/udp/" + +9. References + +9.1. Normative References + + [ABNF] Crocker, D., Ed. and P. Overell, "Augmented BNF for Syntax + Specifications: ABNF", RFC 2234, DOI 10.17487/RFC2234, + November 1997, <https://www.rfc-editor.org/info/rfc2234>. + + [ECN] Ramakrishnan, K., Floyd, S., and D. Black, "The Addition + of Explicit Congestion Notification (ECN) to IP", + RFC 3168, DOI 10.17487/RFC3168, September 2001, + <https://www.rfc-editor.org/info/rfc3168>. + + [EXT-CONNECT2] + McManus, P., "Bootstrapping WebSockets with HTTP/2", + RFC 8441, DOI 10.17487/RFC8441, September 2018, + <https://www.rfc-editor.org/info/rfc8441>. + + [EXT-CONNECT3] + Hamilton, R., "Bootstrapping WebSockets with HTTP/3", + RFC 9220, DOI 10.17487/RFC9220, June 2022, + <https://www.rfc-editor.org/info/rfc9220>. + + [HTTP] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP Semantics", STD 97, RFC 9110, + DOI 10.17487/RFC9110, June 2022, + <https://www.rfc-editor.org/info/rfc9110>. + + [HTTP-DGRAM] + Schinazi, D. and L. Pardue, "HTTP Datagrams and the + Capsule Protocol", RFC 9297, DOI 10.17487/RFC9297, August + 2022, <https://www.rfc-editor.org/info/rfc9297>. + + [HTTP/1.1] Fielding, R., Ed., Nottingham, M., Ed., and J. Reschke, + Ed., "HTTP/1.1", STD 99, RFC 9112, DOI 10.17487/RFC9112, + June 2022, <https://www.rfc-editor.org/info/rfc9112>. + + [HTTP/2] Thomson, M., Ed. and C. Benfield, Ed., "HTTP/2", RFC 9113, + DOI 10.17487/RFC9113, June 2022, + <https://www.rfc-editor.org/info/rfc9113>. + + [HTTP/3] Bishop, M., Ed., "HTTP/3", RFC 9114, DOI 10.17487/RFC9114, + June 2022, <https://www.rfc-editor.org/info/rfc9114>. + + [PROXY-STATUS] + Nottingham, M. and P. Sikora, "The Proxy-Status HTTP + Response Header Field", RFC 9209, DOI 10.17487/RFC9209, + June 2022, <https://www.rfc-editor.org/info/rfc9209>. + + [QUIC] Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [QUIC-DGRAM] + Pauly, T., Kinnear, E., and D. Schinazi, "An Unreliable + Datagram Extension to QUIC", RFC 9221, + DOI 10.17487/RFC9221, March 2022, + <https://www.rfc-editor.org/info/rfc9221>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + + [TCP] Eddy, W., Ed., "Transmission Control Protocol (TCP)", + STD 7, RFC 9293, DOI 10.17487/RFC9293, August 2022, + <https://www.rfc-editor.org/info/rfc9293>. + + [TEMPLATE] Gregorio, J., Fielding, R., Hadley, M., Nottingham, M., + and D. Orchard, "URI Template", RFC 6570, + DOI 10.17487/RFC6570, March 2012, + <https://www.rfc-editor.org/info/rfc6570>. + + [UDP] Postel, J., "User Datagram Protocol", STD 6, RFC 768, + DOI 10.17487/RFC0768, August 1980, + <https://www.rfc-editor.org/info/rfc768>. + + [URI] Berners-Lee, T., Fielding, R., and L. Masinter, "Uniform + Resource Identifier (URI): Generic Syntax", STD 66, + RFC 3986, DOI 10.17487/RFC3986, January 2005, + <https://www.rfc-editor.org/info/rfc3986>. + +9.2. Informative References + + [BEHAVE] Audet, F., Ed. and C. Jennings, "Network Address + Translation (NAT) Behavioral Requirements for Unicast + UDP", BCP 127, RFC 4787, DOI 10.17487/RFC4787, January + 2007, <https://www.rfc-editor.org/info/rfc4787>. + + [DPLPMTUD] Fairhurst, G., Jones, T., Tรผxen, M., Rรผngeler, I., and T. + Vรถlker, "Packetization Layer Path MTU Discovery for + Datagram Transports", RFC 8899, DOI 10.17487/RFC8899, + September 2020, <https://www.rfc-editor.org/info/rfc8899>. + + [ECN-TUNNEL] + Briscoe, B., "Tunnelling of Explicit Congestion + Notification", RFC 6040, DOI 10.17487/RFC6040, November + 2010, <https://www.rfc-editor.org/info/rfc6040>. + + [HELIUM] Schwartz, B. M., "Hybrid Encapsulation Layer for IP and + UDP Messages (HELIUM)", Work in Progress, Internet-Draft, + draft-schwartz-httpbis-helium-00, 25 June 2018, + <https://datatracker.ietf.org/doc/html/draft-schwartz- + httpbis-helium-00>. + + [HiNT] Pardue, L., "HTTP-initiated Network Tunnelling (HiNT)", + Work in Progress, Internet-Draft, draft-pardue-httpbis- + http-network-tunnelling-00, 2 July 2018, + <https://datatracker.ietf.org/doc/html/draft-pardue- + httpbis-http-network-tunnelling-00>. + + [ICMP6] Conta, A., Deering, S., and M. Gupta, Ed., "Internet + Control Message Protocol (ICMPv6) for the Internet + Protocol Version 6 (IPv6) Specification", STD 89, + RFC 4443, DOI 10.17487/RFC4443, March 2006, + <https://www.rfc-editor.org/info/rfc4443>. + + [MASQUE-ORIGINAL] + Schinazi, D., "The MASQUE Protocol", Work in Progress, + Internet-Draft, draft-schinazi-masque-00, 28 February + 2019, <https://datatracker.ietf.org/doc/html/draft- + schinazi-masque-00>. + + [TUNNEL-SECURITY] + Krishnan, S., Thaler, D., and J. Hoagland, "Security + Concerns with IP Tunneling", RFC 6169, + DOI 10.17487/RFC6169, April 2011, + <https://www.rfc-editor.org/info/rfc6169>. + + [UDP-USAGE] + Eggert, L., Fairhurst, G., and G. Shepherd, "UDP Usage + Guidelines", BCP 145, RFC 8085, DOI 10.17487/RFC8085, + March 2017, <https://www.rfc-editor.org/info/rfc8085>. + + [WEBSOCKET] + Fette, I. and A. Melnikov, "The WebSocket Protocol", + RFC 6455, DOI 10.17487/RFC6455, December 2011, + <https://www.rfc-editor.org/info/rfc6455>. + +Acknowledgments + + This document is a product of the MASQUE Working Group, and the + author thanks all MASQUE enthusiasts for their contributions. This + proposal was inspired directly or indirectly by prior work from many + people, in particular [HELIUM] by Ben Schwartz, [HiNT] by Lucas + Pardue, and the original MASQUE Protocol [MASQUE-ORIGINAL] by the + author of this document. + + The author would like to thank Eric Rescorla for suggesting the use + of an HTTP method to proxy UDP. The author is indebted to Mark + Nottingham and Lucas Pardue for the many improvements they + contributed to this document. The extensibility design in this + document came out of the HTTP Datagrams Design Team, whose members + were Alan Frindell, Alex Chernyakhovsky, Ben Schwartz, Eric Rescorla, + Lucas Pardue, Marcus Ihlar, Martin Thomson, Mike Bishop, Tommy Pauly, + Victor Vasiliev, and the author of this document. + +Author's Address + + David Schinazi + Google LLC + 1600 Amphitheatre Parkway + Mountain View, CA 94043 + United States of America + Email: dschinazi.ietf@gmail.com diff --git a/eval/corpora/rfc/RFC 9308 - Applicability of the QUIC Transport Protocol.txt b/eval/corpora/rfc/RFC 9308 - Applicability of the QUIC Transport Protocol.txt new file mode 100644 index 00000000..dd18b4b6 --- /dev/null +++ b/eval/corpora/rfc/RFC 9308 - Applicability of the QUIC Transport Protocol.txt @@ -0,0 +1,1234 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) M. Kรผhlewind +Request for Comments: 9308 Ericsson +Category: Informational B. Trammell +ISSN: 2070-1721 Google Switzerland GmbH + September 2022 + + + Applicability of the QUIC Transport Protocol + +Abstract + + This document discusses the applicability of the QUIC transport + protocol, focusing on caveats impacting application protocol + development and deployment over QUIC. Its intended audience is + designers of application protocol mappings to QUIC and implementors + of these application protocols. + +Status of This Memo + + This document is not an Internet Standards Track specification; it is + published for informational purposes. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Not all documents + approved by the IESG are candidates for any level of Internet + Standard; see Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9308. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 2. The Necessity of Fallback + 3. 0-RTT + 3.1. Replay Attacks + 3.2. Session Resumption versus Keep-Alive + 4. Use of Streams + 4.1. Stream versus Flow Multiplexing + 4.2. Prioritization + 4.3. Ordered and Reliable Delivery + 4.4. Flow Control Deadlocks + 4.5. Stream Limit Commitments + 5. Packetization and Latency + 6. Error Handling + 7. Acknowledgment Efficiency + 8. Port Selection and Application Endpoint Discovery + 8.1. Source Port Selection + 9. Connection Migration + 10. Connection Termination + 11. Information Exposure and the Connection ID + 11.1. Server-Generated Connection ID + 11.2. Mitigating Timing Linkability with Connection ID Migration + 11.3. Using Server Retry for Redirection + 12. Quality of Service (QoS) and Diffserv Code Point (DSCP) + 13. Use of Versions and Cryptographic Handshake + 14. Enabling Deployment of New Versions + 15. Unreliable Datagram Service over QUIC + 16. IANA Considerations + 17. Security Considerations + 18. References + 18.1. Normative References + 18.2. Informative References + Acknowledgments + Contributors + Authors' Addresses + +1. Introduction + + QUIC [QUIC] is a new transport protocol providing a number of + advanced features. While initially designed for the HTTP use case, + it provides capabilities that can be used with a much wider variety + of applications. QUIC is encapsulated in UDP. QUIC version 1 + integrates TLS 1.3 [TLS13] to encrypt all payload data and most + control information. The version of HTTP that uses QUIC is known as + HTTP/3 [QUIC-HTTP]. + + This document provides guidance for application developers who want + to use the QUIC protocol without implementing it on their own. This + includes general guidance for applications operating over HTTP/3 or + directly over QUIC. + + In the following sections, we discuss specific caveats to QUIC's + applicability and issues that application developers must consider + when using QUIC as a transport for their applications. + +2. The Necessity of Fallback + + QUIC uses UDP as a substrate. This enables userspace implementation + and permits traversal of network middleboxes (including NAT) without + requiring updates to existing network infrastructure. + + Measurement studies have shown between 3% [Trammell16] and 5% + [Swett16] of networks block all UDP traffic, though there is little + evidence of other forms of systematic disadvantage to UDP traffic + compared to TCP [Edeline16]. This blocking implies that all + applications running on top of QUIC must either be prepared to accept + connectivity failure on such networks or be engineered to fall back + to some other transport protocol. In the case of HTTP, this fallback + is TLS over TCP. + + The IETF Transport Services (TAPS) specifications [TAPS-ARCH] + describe a system with a common API for multiple protocols. This is + particularly relevant for QUIC as it addresses the implications of + fallback among multiple protocols. + + Specifically, fallback to insecure protocols or to weaker versions of + secure protocols needs to be avoided. In general, an application + that implements fallback needs to consider the security consequences. + A fallback to TCP and TLS exposes control information to modification + and manipulation in the network. Additionally, downgrades to TLS + versions older than 1.3, which is used in QUIC version 1, might + result in significantly weaker cryptographic protection. For + example, the results of protocol negotiation [RFC7301] only have + confidentiality protection if TLS 1.3 is used. + + These applications must operate, perhaps with impaired functionality, + in the absence of features provided by QUIC not present in the + fallback protocol. For fallback to TLS over TCP, the most obvious + difference is that TCP does not provide stream multiplexing, and + therefore stream multiplexing would need to be implemented in the + application layer if needed. Further, TCP implementations and + network paths often do not support the TCP Fast Open (TFO) option + [RFC7413], which enables sending of payload data together with the + first control packet of a new connection as also provided by 0-RTT + session resumption in QUIC. Note that there is some evidence of + middleboxes blocking SYN data even if TFO was successfully negotiated + (see [PaaschNanog]). And even if Fast Open successfully operates end + to end, it is limited to a single packet of TLS handshake and + application data, unlike QUIC 0-RTT. + + Moreover, while encryption (in this case TLS) is inseparably + integrated with QUIC, TLS negotiation over TCP can be blocked. If + TLS over TCP cannot be supported, the connection should be aborted, + and the application then ought to present a suitable prompt to the + user that secure communication is unavailable. + + In summary, any fallback mechanism is likely to impose a degradation + of performance and can degrade security; however, fallback must not + silently violate the application's expectation of confidentiality or + integrity of its payload data. + +3. 0-RTT + + QUIC provides for 0-RTT connection establishment. Though the same + facility exists in TLS 1.3 with TCP, 0-RTT presents opportunities and + challenges for applications using QUIC. + + A transport protocol that provides 0-RTT connection establishment is + qualitatively different from one that does not provide 0-RTT from the + point of view of the application using it. Relative trade-offs + between the cost of closing and reopening a connection and trying to + keep it open are different; see Section 3.2. + + An application needs to deliberately choose to use 0-RTT, as 0-RTT + carries a risk of replay attack. Application protocols that use + 0-RTT require a profile that describes the types of information that + can be safely sent. For HTTP, this profile is described in + [HTTP-REPLAY]. + +3.1. Replay Attacks + + Retransmission or malicious replay of data contained in 0-RTT packets + could cause the server side to receive multiple copies of the same + data. + + Application data sent by the client in 0-RTT packets could be + processed more than once if it is replayed. Applications need to be + aware of what is safe to send in 0-RTT. Application protocols that + seek to enable the use of 0-RTT need a careful analysis and a + description of what can be sent in 0-RTT; see Section 5.6 of + [QUIC-TLS]. + + In some cases, it might be sufficient to limit application data sent + in 0-RTT to data that does not cause actions with lasting effects at + a server. Initiating data retrieval or establishing configuration + are examples of actions that could be safe. Idempotent operations -- + those for which repetition has the same net effect as a single + operation -- might be safe. However, it is also possible to combine + individually idempotent operations into a non-idempotent sequence of + operations. + + Once a server accepts 0-RTT data, there is no means of selectively + discarding data that is received. However, protocols can define ways + to reject individual actions that might be unsafe if replayed. + + Some TLS implementations and deployments might be able to provide + partial or even complete replay protection, which could be used to + manage replay risk. + +3.2. Session Resumption versus Keep-Alive + + Because QUIC is encapsulated in UDP, applications using QUIC must + deal with short network idle timeouts. Deployed stateful middleboxes + will generally establish state for UDP flows on the first packet sent + and keep state for much shorter idle periods than for TCP. [RFC5382] + suggests a TCP idle period of at least 124 minutes, though there is + no evidence of widespread implementation of this guideline in the + literature. However, short network timeout for UDP is well- + documented. According to a 2010 study ([Hatonen10]), UDP + applications can assume that any NAT binding or other state entry can + expire after just thirty seconds of inactivity. Section 3.5 of + [RFC8085] further discusses keep-alive intervals for UDP: it requires + that there is a minimum value of 15 seconds, but recommends larger + values, or that keep-alive is omitted entirely. + + By using a connection ID, QUIC is designed to be robust to NAT + rebinding after a timeout. However, this only helps if one endpoint + maintains availability at the address its peer uses and the peer is + the one to send after the timeout occurs. + + Some QUIC connections might not be robust to NAT rebinding because + the routing infrastructure (in particular, load balancers) uses the + address/port 4-tuple to direct traffic. Furthermore, middleboxes + with functions other than address translation could still affect the + path. In particular, some firewalls do not admit server traffic for + which the firewall has no recent state for a corresponding packet + sent from the client. + + QUIC applications can adjust idle periods to manage the risk of + timeout. Idle periods and the network idle timeout are distinct from + the connection idle timeout, which is defined as the minimum of + either endpoint's idle timeout parameter; see Section 10.1 of [QUIC]. + There are three options: + + * Ignore the issue if the application-layer protocol consists only + of interactions with no or very short idle periods or if the + protocol's resistance to NAT rebinding is sufficient. + + * Ensure there are no long idle periods. + + * Resume the session after a long idle period, using 0-RTT + resumption when appropriate. + + The first strategy is the easiest, but it only applies to certain + applications. + + Either the server or the client in a QUIC application can send PING + frames as keep-alives to prevent the connection and any on-path state + from timing out. Recommendations for the use of keep-alives are + application specific, mainly depending on the latency requirements + and message frequency of the application. In this case, the + application mapping must specify whether the client or server is + responsible for keeping the application alive. While [Hatonen10] + suggests that 30 seconds might be a suitable value for the public + Internet when a NAT is on path, larger values are preferable if the + deployment can consistently survive NAT rebinding or is known to be + in a controlled environment (e.g., data centers) in order to lower + network and computational load. + + Sending PING frames more frequently than every 30 seconds over long + idle periods may result in excessive unproductive traffic in some + situations and unacceptable power usage for power-constrained + (mobile) devices. Additionally, timeouts shorter than 30 seconds can + make it harder to handle transient network interruptions, such as + Virtual Machine (VM) migration or coverage loss during mobility. See + [RFC8085], especially Section 3.5. + + Alternatively, the client (but not the server) can use session + resumption instead of sending keep-alive traffic. In this case, a + client that wants to send data to a server over a connection that has + been idle longer than the server's idle timeout (available from the + idle_timeout transport parameter) can simply reconnect. When + possible, this reconnection can use 0-RTT session resumption, + reducing the latency involved with restarting the connection. Of + course, this approach is only valid in cases in which it is safe to + use 0-RTT and when the client is the restarting peer. + + The trade-offs between resumption and keep-alives need to be + evaluated on a per-application basis. In general, applications + should use keep-alives only in circumstances where continued + communication is highly likely; [QUIC-HTTP], for instance, recommends + using keep-alives only when a request is outstanding. + +4. Use of Streams + + QUIC's stream multiplexing feature allows applications to run + multiple streams over a single connection without head-of-line + blocking between streams. Stream data is carried within frames where + one QUIC packet on the wire can carry one or multiple stream frames. + + Streams can be unidirectional or bidirectional, and a stream may be + initiated either by client or server. Only the initiator of a + unidirectional stream can send data on it. + + Streams and connections can each carry a maximum of 2^62-1 bytes in + each direction due to encoding limitations on stream offsets and + connection flow control limits. In the presently unlikely event that + this limit is reached by an application, a new connection would need + to be established. + + Streams can be independently opened and closed, gracefully or + abruptly. An application can gracefully close the egress direction + of a stream by instructing QUIC to send a FIN bit in a STREAM frame. + It cannot gracefully close the ingress direction without a peer- + generated FIN, much like in TCP. However, an endpoint can abruptly + close the egress direction or request that its peer abruptly close + the ingress direction; these actions are fully independent of each + other. + + QUIC does not provide an interface for exceptional handling of any + stream. If a stream that is critical for an application is closed, + the application can generate error messages on the application layer + to inform the other end and/or the higher layer, which can eventually + terminate the QUIC connection. + + Mapping of application data to streams is application specific and + described for HTTP/3 in [QUIC-HTTP]. There are a few general + principles to apply when designing an application's use of streams: + + * A single stream provides ordering. If the application requires + certain data to be received in order, that data should be sent on + the same stream. There is no guarantee of transmission, + reception, or delivery order across streams. + + * Multiple streams provide concurrency. Data that can be processed + independently, and therefore would suffer from head-of-line + blocking if forced to be received in order, should be transmitted + over separate streams. + + * Streams can provide message orientation and allow messages to be + canceled. If one message is mapped to a single stream, resetting + the stream to expire an unacknowledged message can be used to + emulate partial reliability for that message. + + If a QUIC receiver has opened the maximum allowed concurrent streams, + and the sender indicates that more streams are needed, it does not + automatically lead to an increase of the maximum number of streams by + the receiver. Therefore, an application should consider the maximum + number of allowed, currently open, and currently used streams when + determining how to map data to streams. + + QUIC assigns a numerical identifier, called the stream ID, to each + stream. While the relationship between these identifiers and stream + types is clearly defined in version 1 of QUIC, future versions might + change this relationship for various reasons. QUIC implementations + should expose the properties of each stream (which endpoint initiated + the stream, whether the stream is unidirectional or bidirectional, + the stream ID used for the stream); applications should query for + these properties rather than attempting to infer them from the stream + ID. + + The method of allocating stream identifiers to streams opened by the + application might vary between transport implementations. Therefore, + an application should not assume a particular stream ID will be + assigned to a stream that has not yet been allocated. For example, + HTTP/3 uses stream IDs to refer to streams that have already been + opened but makes no assumptions about future stream IDs or the way in + which they are assigned (see Section 6 of [QUIC-HTTP]). + +4.1. Stream versus Flow Multiplexing + + Streams are meaningful only to the application; since stream + information is carried inside QUIC's encryption boundary, a given + packet exposes no information about which stream(s) are carried + within the packet. Therefore, stream multiplexing is not intended to + be used for differentiating streams in terms of network treatment. + Application traffic requiring different network treatment should + therefore be carried over different 5-tuples (i.e., multiple QUIC + connections). Given QUIC's ability to send application data in the + first RTT of a connection (if a previous connection to the same host + has been successfully established to provide the necessary + credentials), the cost of establishing another connection is + extremely low. + +4.2. Prioritization + + Stream prioritization is not exposed to either the network or the + receiver. Prioritization is managed by the sender, and the QUIC + transport should provide an interface for applications to prioritize + streams [QUIC]. Applications can implement their own prioritization + scheme on top of QUIC: an application protocol that runs on top of + QUIC can define explicit messages for signaling priority, such as + those defined in [RFC9218] for HTTP. An application protocol can + define rules that allow an endpoint to determine priority based on + context or can provide a higher-level interface and leave the + determination to the application on top. + + Priority handling of retransmissions can be implemented by the sender + in the transport layer. [QUIC] recommends retransmitting lost data + before new data, unless indicated differently by the application. + When a QUIC endpoint uses fully reliable streams for transmission, + prioritization of retransmissions will be beneficial in most cases, + filling in gaps and freeing up the flow control window. For + partially reliable or unreliable streams, priority scheduling of + retransmissions over data of higher-priority streams might not be + desirable. For such streams, QUIC could either provide an explicit + interface to control prioritization or derive the prioritization + decision from the reliability level of the stream. + +4.3. Ordered and Reliable Delivery + + QUIC streams enable ordered and reliable delivery. Though it is + possible for an implementation to provide options that use streams + for partial reliability or out-of-order delivery, most + implementations will assume that data is reliably delivered in order. + + Under this assumption, an endpoint that receives stream data might + not make forward progress until data that is contiguous with the + start of a stream is available. In particular, a receiver might + withhold flow control credit until contiguous data is delivered to + the application; see Section 2.2 of [QUIC]. To support this receive + logic, an endpoint will send stream data until it is acknowledged, + ensuring that data at the start of the stream is sent and + acknowledged first. + + An endpoint that uses a different sending behavior and does not + negotiate that change with its peer might encounter performance + issues or deadlocks. + +4.4. Flow Control Deadlocks + + QUIC flow control (Section 4 of [QUIC]) provides a means of managing + access to the limited buffers that endpoints have for incoming data. + This mechanism limits the amount of data that can be in buffers in + endpoints or in transit on the network. However, there are several + ways in which limits can produce conditions that can cause a + connection to either perform suboptimally or become deadlocked. + + Deadlocks in flow control are possible for any protocol that uses + QUIC, though whether they become a problem depends on how + implementations consume data and provide flow control credit. + Understanding what causes deadlocking might help implementations + avoid deadlocks. + + The size and rate of updates to flow control credit can affect + performance. Applications that use QUIC often have a data consumer + that reads data from transport buffers. Some implementations might + have independent receive buffers at the transport layer and + application layer. Consuming data does not always imply it is + immediately processed. However, a common implementation technique is + to extend flow control credit to the sender by emitting MAX_DATA and/ + or MAX_STREAM_DATA frames as data is consumed. Delivery of these + frames is affected by the latency of the back channel from the + receiver to the data sender. If credit is not extended in a timely + manner, the sending application can be blocked, effectively + throttling the sender. + + Large application messages can produce deadlocking if the recipient + does not read data from the transport incrementally. If the message + is larger than the flow control credit available and the recipient + does not release additional flow control credit until the entire + message is received and delivered, a deadlock can occur. This is + possible even where stream flow control limits are not reached + because connection flow control limits can be consumed by other + streams. + + A length-prefixed message format makes it easier for a data consumer + to leave data unread in the transport buffer and thereby withhold + flow control credit. If flow control limits prevent the remainder of + a message from being sent, a deadlock will result. A length prefix + might also enable the detection of this sort of deadlock. Where + application protocols have messages that might be processed as a + single unit, reserving flow control credit for the entire message + atomically makes this style of deadlock less likely. + + A data consumer can eagerly read all data as it becomes available in + order to make the receiver extend flow control credit and reduce the + chances of a deadlock. However, such a data consumer might need + other means for holding a peer accountable for the additional state + it keeps for partially processed messages. + + Deadlocking can also occur if data on different streams is + interdependent. Suppose that data on one stream arrives before the + data on a second stream on which it depends. A deadlock can occur if + the first stream is left unread, preventing the receiver from + extending flow control credit for the second stream. To reduce the + likelihood of deadlock for interdependent data, the sender should + ensure that dependent data is not sent until the data it depends on + has been accounted for in both stream- and connection-level flow + control credit. + + Some deadlocking scenarios might be resolved by canceling affected + streams with STOP_SENDING or RESET_STREAM. Canceling some streams + results in the connection being terminated in some protocols. + +4.5. Stream Limit Commitments + + QUIC endpoints are responsible for communicating the cumulative limit + of streams they would allow to be opened by their peer. Initial + limits are advertised using the initial_max_streams_bidi and + initial_max_streams_uni transport parameters. As streams are opened + and closed, they are consumed, and the cumulative total is + incremented. Limits can be increased using the MAX_STREAMS frame, + but there is no mechanism to reduce limits. Once stream limits are + reached, no more streams can be opened, which prevents applications + using QUIC from making further progress. At this stage, connections + can be terminated via idle timeout or explicit close; see Section 10. + + An application that uses QUIC and communicates a cumulative stream + limit might require the connection to be closed before the limit is + reached, e.g., to stop the server in order to perform scheduled + maintenance. Immediate connection close causes abrupt closure of + actively used streams. Depending on how an application uses QUIC + streams, this could be undesirable or detrimental to behavior or + performance. + + A more graceful closure technique is to stop sending increases to + stream limits and allow the connection to naturally terminate once + remaining streams are consumed. However, the period of time it takes + to do so is dependent on the peer, and an unpredictable closing + period might not fit application or operational needs. Applications + using QUIC can be conservative with open stream limits in order to + reduce the commitment and indeterminism. However, being overly + conservative with stream limits affects stream concurrency. + Balancing these aspects can be specific to applications and their + deployments. + + Instead of relying on stream limits to avoid abrupt closure, an + application layer's graceful close mechanism can be used to + communicate the intention to explicitly close the connection at some + future point. HTTP/3 provides such a mechanism using the GOAWAY + frame. In HTTP/3, when the GOAWAY frame is received by a client, it + stops opening new streams even if the cumulative stream limit would + allow. Instead, the client would create a new connection on which to + open further streams. Once all streams are closed on the old + connection, it can be terminated safely by a connection close or + after expiration of the idle timeout (see Section 10). + +5. Packetization and Latency + + QUIC exposes an interface that provides multiple streams to the + application; however, the application usually cannot control how data + transmitted over those streams is mapped into frames or how those + frames are bundled into packets. + + By default, many implementations will try to pack STREAM frames from + one or more streams into each QUIC packet, in order to minimize + bandwidth consumption and computational costs (see Section 13 of + [QUIC]). If there is not enough data available to fill a packet, an + implementation might wait for a short time to optimize bandwidth + efficiency instead of latency. This delay can either be + preconfigured or dynamically adjusted based on the observed sending + pattern of the application. + + If the application requires low latency, with only small chunks of + data to send, it may be valuable to indicate to QUIC that all data + should be sent out immediately. Alternatively, if the application + expects to use a specific sending pattern, it can also provide a + suggested delay to QUIC for how long to wait before bundling frames + into a packet. + + Similarly, an application usually has no control over the length of a + QUIC packet on the wire. QUIC provides the ability to add a PADDING + frame to arbitrarily increase the size of packets. Padding is used + by QUIC to ensure that the path is capable of transferring datagrams + of at least a certain size during the handshake (see Sections 8.1 and + 14.1 of [QUIC]) and for path validation after connection migration + (see Section 8.2 of [QUIC]) as well as for Datagram Packetization + Layer PMTU Discovery (DPLPMTUD) (see Section 14.3 of [QUIC]). + + Padding can also be used by an application to reduce leakage of + information about the data that is sent. A QUIC implementation can + expose an interface that allows an application layer to specify how + to apply padding. + +6. Error Handling + + QUIC recommends that endpoints signal any detected errors to the + peer. Errors can occur at the transport layer and the application + layer. Transport errors, such as a protocol violation, affect the + entire connection. Applications that use QUIC can define their own + error detection and signaling (see, for example, Section 8 of + [QUIC-HTTP]). Application errors can affect an entire connection or + a single stream. + + QUIC defines an error code space that is used for error handling at + the transport layer. QUIC encourages endpoints to use the most + specific code, although any applicable code is permitted, including + generic ones. + + Applications using QUIC define an error code space that is + independent of QUIC or other applications (see, for example, + Section 8.1 of [QUIC-HTTP]). The values in an application error code + space can be reused across connection-level and stream-level errors. + + Connection errors lead to connection termination. They are signaled + using a CONNECTION_CLOSE frame, which contains an error code and a + reason field that can be zero length. Different types of + CONNECTION_CLOSE frames are used to signal transport and application + errors. + + Stream errors lead to stream termination. These are signaled using + STOP_SENDING or RESET_STREAM frames, which contain only an error + code. + +7. Acknowledgment Efficiency + + QUIC version 1 without extensions uses an acknowledgment strategy + adopted from TCP (see Section 13.2 of [QUIC]). That is, it + recommends that every other packet is acknowledged. However, + generating and processing QUIC acknowledgments consumes resources at + a sender and receiver. Acknowledgments also incur forwarding costs + and contribute to link utilization, which can impact performance over + some types of network. Applications might be able to improve overall + performance by using alternative strategies that reduce the rate of + acknowledgments. [QUIC-ACK-FREQUENCY] describes an extension to + signal the desired delay of acknowledgments and discusses use cases + as well as implications for congestion control and recovery. + +8. Port Selection and Application Endpoint Discovery + + In general, port numbers serve two purposes: "first, they provide a + demultiplexing identifier to differentiate transport sessions between + the same pair of endpoints, and second, they may also identify the + application protocol and associated service to which processes + connect" (Section 3 of [RFC6335]). The assumption that an + application can be identified in the network based on the port number + is less true today due to encapsulation and mechanisms for dynamic + port assignments, as noted in [RFC6335]. + + As QUIC is a general-purpose transport protocol, there are no + requirements that servers use a particular UDP port for QUIC. For an + application with a fallback to TCP that does not already have an + alternate mapping to UDP, it is usually appropriate to register (if + necessary) and use the UDP port number corresponding to the TCP port + already registered for the application. For example, the default + port for HTTP/3 [QUIC-HTTP] is UDP port 443, analogous to HTTP/1.1 or + HTTP/2 over TLS over TCP. + + Given the prevalence of the assumption in network management practice + that a port number maps unambiguously to an application, the use of + ports that cannot easily be mapped to a registered service name might + lead to blocking or other changes to the forwarding behavior by + network elements such as firewalls that use the port number for + application identification. + + Applications could define an alternate endpoint discovery mechanism + to allow the usage of ports other than the default. For example, + HTTP/3 (Sections 3.2 and 3.3 of [QUIC-HTTP]) specifies the use of + HTTP Alternative Services [RFC7838] for an HTTP origin to advertise + the availability of an equivalent HTTP/3 endpoint on a certain UDP + port by using "h3" as the Application-Layer Protocol Negotiation + (ALPN) [RFC7301] token. + + ALPN permits the client and server to negotiate which of several + protocols will be used on a given connection. Therefore, multiple + applications might be supported on a single UDP port based on the + ALPN token offered. Applications using QUIC are required to register + an ALPN token for use in the TLS handshake. + + As QUIC version 1 deferred defining a complete version negotiation + mechanism, HTTP/3 requires QUIC version 1 and defines the ALPN token + ("h3") to only apply to that version. So far, no single approach has + been selected for managing the use of different QUIC versions, + neither in HTTP/3 nor in general. Application protocols that use + QUIC need to consider how the protocol will manage different QUIC + versions. Decisions for those protocols might be informed by choices + made by other protocols, like HTTP/3. + +8.1. Source Port Selection + + Some UDP protocols are vulnerable to reflection attacks, where an + attacker is able to direct traffic to a third party as a denial of + service. For example, these source ports are associated with + applications known to be vulnerable to reflection attacks, often due + to server misconfiguration: + + * port 53 - DNS [RFC1034] + + * port 123 - NTP [RFC5905] + + * port 1900 - SSDP [SSDP] + + * port 5353 - mDNS [RFC6762] + + * port 11211 - memcache + + Services might block source ports associated with protocols known to + be vulnerable to reflection attacks to avoid the overhead of + processing large numbers of packets. However, this practice has + negative effects on clients -- not only does it require establishment + of a new connection but in some instances might cause the client to + avoid using QUIC for that service for a period of time and downgrade + to a non-UDP protocol (see Section 2). + + As a result, client implementations are encouraged to avoid using + source ports associated with protocols known to be vulnerable to + reflection attacks. Note that following the general guidance for + client implementations given in [RFC6335], to use ephemeral ports in + the range 49152-65535, has the effect of avoiding these ports. Note + that other source ports might be reflection vectors as well. + +9. Connection Migration + + QUIC supports connection migration by the client. If the client's IP + address changes, a QUIC endpoint can still associate packets with an + existing transport connection using the Destination Connection ID + field (see Section 11) in the QUIC header. This supports cases where + the address information changes, such as NAT rebinding, the + intentional change of the local interface, the expiration of a + temporary IPv6 address [RFC8981], or the indication from the server + of a preferred address (Section 9.6 of [QUIC]). + + Use of a non-zero-length connection ID for the server is strongly + recommended if any clients are or could be behind a NAT. A non-zero- + length connection ID is also strongly recommended when active + migration is supported. If a connection is intentionally migrated to + a new path, a new connection ID is used to minimize linkability by + network observers. The other QUIC endpoint uses the connection ID to + link different addresses to the same connection and entity if a non- + zero-length connection ID is provided. + + The base specification of QUIC version 1 only supports the use of a + single network path at a time, which enables failover use cases. + Path validation is required so that endpoints validate paths before + use to avoid address spoofing attacks. Path validation takes at + least one RTT, and congestion control will also be reset after path + migration. Therefore, migration usually has a performance impact. + + QUIC probing packets, which can be sent on multiple paths at once, + are used to perform address validation as well as measure path + characteristics. Probing packets cannot carry application data but + likely contain padding frames. Endpoints can use information about + their receipt as input to congestion control for that path. + Applications could use information learned from probing to inform a + decision to switch paths. + + Only the client can actively migrate in version 1 of QUIC. However, + servers can indicate during the handshake that they prefer to + transfer the connection to a different address after the handshake. + For instance, this could be used to move from an address that is + shared by multiple servers to an address that is unique to the server + instance. The server can provide an IPv4 and an IPv6 address in a + transport parameter during the TLS handshake, and the client can + select between the two if both are provided. See Section 9.6 of + [QUIC]. + +10. Connection Termination + + QUIC connections are terminated in one of three ways: implicit idle + timeout, explicit immediate close, or explicit stateless reset. + + QUIC does not provide any mechanism for graceful connection + termination; applications using QUIC can define their own graceful + termination process (see, for example, Section 5.2 of [QUIC-HTTP]). + + QUIC idle timeout is enabled via transport parameters. The client + and server announce a timeout period, and the effective value for the + connection is the minimum of the two values. After the timeout + period elapses, the connection is silently closed. An application + therefore should be able to configure its own maximum value, as well + as have access to the computed minimum value for this connection. An + application may adjust the maximum idle timeout for new connections + based on the number of open or expected connections since shorter + timeout values may free up resources more quickly. + + Application data exchanged on streams or in datagrams defers the QUIC + idle timeout. Applications that provide their own keep-alive + mechanisms will therefore keep a QUIC connection alive. Applications + that do not provide their own keep-alive can use transport-layer + mechanisms (see Section 10.1.2 of [QUIC] and Section 3.2). However, + QUIC implementation interfaces for controlling such transport + behavior can vary, affecting the robustness of such approaches. + + An immediate close is signaled by a CONNECTION_CLOSE frame (see + Section 6). Immediate close causes all streams to become immediately + closed, which may affect applications; see Section 4.5. + + A stateless reset is an option of last resort for an endpoint that + does not have access to connection state. Receiving a stateless + reset is an indication of an unrecoverable error distinct from + connection errors in that there is no application-layer information + provided. + +11. Information Exposure and the Connection ID + + QUIC exposes some information to the network in the unencrypted part + of the header either before the encryption context is established or + because the information is intended to be used by the network. For + more information on manageability of QUIC, see [QUIC-MANAGEABILITY]. + QUIC has a long header that exposes some additional information (the + version and the source connection ID), while the short header exposes + only the destination connection ID. In QUIC version 1, the long + header is used during connection establishment, while the short + header is used for data transmission in an established connection. + + The connection ID can be zero length. Zero-length connection IDs can + be chosen on each endpoint individually and on any packet except the + first packets sent by clients during connection establishment. + + An endpoint that selects a zero-length connection ID will receive + packets with a zero-length destination connection ID. The endpoint + needs to use other information, such as the source and destination IP + address and port number to identify which connection is referred to. + This could mean that the endpoint is unable to match datagrams to + connections successfully if these values change, making the + connection effectively unable to survive NAT rebinding or migrate to + a new path. + +11.1. Server-Generated Connection ID + + QUIC supports a server-generated connection ID that is transmitted to + the client during connection establishment (see Section 7.2 of + [QUIC]). Servers behind load balancers may need to change the + connection ID during the handshake, encoding the identity of the + server or information about its load balancing pool, in order to + support stateless load balancing. + + Server deployments with load balancers and other routing + infrastructure need to ensure that this infrastructure consistently + routes packets to the server instance that has the connection state, + even if addresses, ports, or connection IDs change. This might + require coordination between servers and infrastructure. One method + of achieving this involves encoding routing information into the + connection ID. For an example of this technique, see [QUIC-LB]. + +11.2. Mitigating Timing Linkability with Connection ID Migration + + If QUIC endpoints do not issue fresh connection IDs, then clients + cannot reduce the linkability of address migration by using them. + Choosing values that are unlinkable to an outside observer ensures + that activity on different paths cannot be trivially correlated using + the connection ID. + + While sufficiently robust connection ID generation schemes will + mitigate linkability issues, they do not provide full protection. + Analysis of the lifetimes of 6-tuples (source and destination + addresses as well as the migrated Connection ID) may expose these + links anyway. + + In the case where connection migration in a server pool is rare, it + is trivial for an observer to associate two connection IDs. + Conversely, where every server handles multiple simultaneous + migrations, even an exposed server mapping may be insufficient + information. + + The most efficient mitigations for these attacks are through network + design and/or operational practices, by using a load-balancing + architecture that loads more flows onto a single server-side address, + by coordinating the timing of migrations in an attempt to increase + the number of simultaneous migrations at a given time, or by using + other means. + +11.3. Using Server Retry for Redirection + + QUIC provides a Retry packet that can be sent by a server in response + to the client Initial packet. The server may choose a new connection + ID in that packet, and the client will retry by sending another + client Initial packet with the server-selected connection ID. This + mechanism can be used to redirect a connection to a different server, + e.g., due to performance reasons or when servers in a server pool are + upgraded gradually and therefore may support different versions of + QUIC. + + In this case, it is assumed that all servers belonging to a certain + pool are served in cooperation with load balancers that forward the + traffic based on the connection ID. A server can choose the + connection ID in the Retry packet such that the load balancer will + redirect the next Initial packet to a different server in that pool. + Alternatively, the load balancer can directly offer a Retry offload + as further described in [QUIC-RETRY]. + + The approach described in Section 4 of [RFC5077] for constructing TLS + resumption tickets provides an example that can be also applied to + validation tokens. However, the use of more modern cryptographic + algorithms is highly recommended. + +12. Quality of Service (QoS) and Diffserv Code Point (DSCP) + + QUIC, as defined in [QUIC], has a single congestion controller and + recovery handler. This design assumes that all packets of a QUIC + connection, or at least with the same 5-tuple {dest addr, source + addr, protocol, dest port, source port}, that have the same Diffserv + Code Point (DSCP) [RFC2475] will receive similar network treatment + since feedback about loss or delay of each packet is used as input to + the congestion controller. Therefore, packets belonging to the same + connection should use a single DSCP. Section 5.1 of [RFC7657] + provides a discussion of Diffserv interactions with datagram + transport protocols [RFC7657] (in this respect, the interactions with + QUIC resemble those of Stream Control Transmission Protocol (SCTP)). + + When multiplexing multiple flows over a single QUIC connection, the + selected DSCP value should be the one associated with the highest + priority requested for all multiplexed flows. + + If differential network treatment is desired, e.g., by the use of + different DSCPs, multiple QUIC connections to the same server may be + used. In general, it is recommended to minimize the number of QUIC + connections to the same server to avoid increased overhead and, more + importantly, competing congestion control. + + As in other uses of Diffserv, when a packet enters a network segment + that does not support the DSCP value, this could result in the + connection not receiving the network treatment it expects. The DSCP + value in this packet could also be remarked as the packet travels + along the network path, changing the requested treatment. + +13. Use of Versions and Cryptographic Handshake + + Versioning in QUIC may change the protocol's behavior completely, + except for the meaning of a few header fields that have been declared + to be invariant [QUIC-INVARIANTS]. A version of QUIC with a higher + version number will not necessarily provide a better service but + might simply provide a different feature set. As such, an + application needs to be able to select which versions of QUIC it + wants to use. + + A new version could use an encryption scheme other than TLS 1.3 or + higher. [QUIC] specifies requirements for the cryptographic + handshake as currently realized by TLS 1.3 and described in a + separate specification [QUIC-TLS]. This split is performed to enable + lightweight versioning with different cryptographic handshakes. + + The "QUIC Versions" registry established in [QUIC] allows for + provisional registrations for experimentation. Registration, also of + experimental versions, is important to avoid collision. Experimental + versions should not be used long-term or registered as permanent to + minimize the risk of fingerprinting based on the version number. + +14. Enabling Deployment of New Versions + + QUIC version 1 does not specify a version negotiation mechanism in + the base specification, but [QUIC-VERSION-NEGOTIATION] proposes an + extension that provides compatible version negotiation. + + This approach uses a three-stage deployment mechanism, enabling + progressive rollout and experimentation with multiple versions across + a large server deployment. In this approach, all servers in the + deployment must accept connections using a new version (stage 1) + before any server advertises it (stage 2), and authentication of the + new version (stage 3) only proceeds after advertising of that version + is completely deployed. + + See Section 5 of [QUIC-VERSION-NEGOTIATION] for details. + +15. Unreliable Datagram Service over QUIC + + [RFC9221] specifies a QUIC extension to enable sending and receiving + unreliable datagrams over QUIC. Unlike operating directly over UDP, + applications that use the QUIC datagram service do not need to + implement their own congestion control, per [RFC8085], as QUIC + datagrams are congestion controlled. + + QUIC datagrams are not flow controlled, and as such data chunks may + be dropped if the receiver is overloaded. While the reliable + transmission service of QUIC provides a stream-based interface to + send and receive data in order over multiple QUIC streams, the + datagram service has an unordered message-based interface. If + needed, an application-layer framing can be used on top to allow + separate flows of unreliable datagrams to be multiplexed on one QUIC + connection. + +16. IANA Considerations + + This document has no actions for IANA; however, note that Section 8 + recommends that an application that has already registered a TCP port + but wants to specify QUIC as a transport should register a UDP port + analogous to their existing TCP registration. + +17. Security Considerations + + See the security considerations in [QUIC] and [QUIC-TLS]; the + security considerations for the underlying transport protocol are + relevant for applications using QUIC. Considerations on linkability, + replay attacks, and randomness discussed in [QUIC-TLS] should be + taken into account when deploying and using QUIC. + + Further, migration to a new address exposes a linkage between client + addresses to the server and may expose this linkage also to the path + if the connection ID cannot be changed or flows can otherwise be + correlated. When migration is supported, this needs to be considered + with respective to user privacy. + + Application developers should note that any fallback they use when + QUIC cannot be used due to network blocking of UDP should guarantee + the same security properties as QUIC. If this is not possible, the + connection should fail to allow the application to explicitly handle + fallback to a less-secure alternative. See Section 2. + + Further, [QUIC-HTTP] provides security considerations specific to + HTTP. However, discussions such as on cross-protocol attacks, + traffic analysis and padding, or migration might be relevant for + other applications using QUIC as well. + +18. References + +18.1. Normative References + + [QUIC] Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [QUIC-INVARIANTS] + Thomson, M., "Version-Independent Properties of QUIC", + RFC 8999, DOI 10.17487/RFC8999, May 2021, + <https://www.rfc-editor.org/info/rfc8999>. + + [QUIC-TLS] Thomson, M., Ed. and S. Turner, Ed., "Using TLS to Secure + QUIC", RFC 9001, DOI 10.17487/RFC9001, May 2021, + <https://www.rfc-editor.org/info/rfc9001>. + +18.2. Informative References + + [Edeline16] + Edeline, K., Kรผhlewind, M., Trammell, B., Aben, E., and B. + Donnet, "Using UDP for Internet Transport Evolution", + DOI 10.48550/arXiv.1612.07816, 22 December 2016, + <https://arxiv.org/abs/1612.07816>. + + [Hatonen10] + Hรคtรถnen, S., Nyrhinen, A., Eggert, L., Strowes, S., + Sarolahti, P., and M. Kojo, "An Experimental Study of Home + Gateway Characteristics", Proc. ACM IMC 2010, November + 2010, <https://conferences.sigcomm.org/imc/2010/papers/ + p260.pdf>. + + [HTTP-REPLAY] + Thomson, M., Nottingham, M., and W. Tarreau, "Using Early + Data in HTTP", RFC 8470, DOI 10.17487/RFC8470, September + 2018, <https://www.rfc-editor.org/info/rfc8470>. + + [PaaschNanog] + Paasch, C., "Network support for TCP Fast Open", NANOG 67 + Presentation, 13 June 2016, + <https://www.nanog.org/sites/default/files/ + Paasch_Network_Support.pdf>. + + [QUIC-ACK-FREQUENCY] + Iyengar, J. and I. Swett, "QUIC Acknowledgement + Frequency", Work in Progress, Internet-Draft, draft-ietf- + quic-ack-frequency-02, 11 July 2022, + <https://datatracker.ietf.org/doc/html/draft-ietf-quic- + ack-frequency-02>. + + [QUIC-HTTP] + Bishop, M., Ed., "HTTP/3", RFC 9114, DOI 10.17487/RFC9114, + June 2022, <https://www.rfc-editor.org/info/rfc9114>. + + [QUIC-LB] Duke, M., Banks, N., and C. Huitema, "QUIC-LB: Generating + Routable QUIC Connection IDs", Work in Progress, Internet- + Draft, draft-ietf-quic-load-balancers-14, 11 July 2022, + <https://datatracker.ietf.org/doc/html/draft-ietf-quic- + load-balancers-14>. + + [QUIC-MANAGEABILITY] + Kรผhlewind, M. and B. Trammell, "Manageability of the QUIC + Transport Protocol", RFC 9312, DOI 10.17487/RFC9312, + September 2022, <https://www.rfc-editor.org/info/rfc9312>. + + [QUIC-RETRY] + Duke, M. and N. Banks, "QUIC Retry Offload", Work in + Progress, Internet-Draft, draft-ietf-quic-retry-offload- + 00, 25 May 2022, <https://datatracker.ietf.org/doc/html/ + draft-ietf-quic-retry-offload-00>. + + [QUIC-VERSION-NEGOTIATION] + Schinazi, D. and E. Rescorla, "Compatible Version + Negotiation for QUIC", Work in Progress, Internet-Draft, + draft-ietf-quic-version-negotiation-10, 27 September 2022, + <https://datatracker.ietf.org/doc/html/draft-ietf-quic- + version-negotiation-10>. + + [RFC1034] Mockapetris, P., "Domain names - concepts and facilities", + STD 13, RFC 1034, DOI 10.17487/RFC1034, November 1987, + <https://www.rfc-editor.org/info/rfc1034>. + + [RFC2475] Blake, S., Black, D., Carlson, M., Davies, E., Wang, Z., + and W. Weiss, "An Architecture for Differentiated + Services", RFC 2475, DOI 10.17487/RFC2475, December 1998, + <https://www.rfc-editor.org/info/rfc2475>. + + [RFC5077] Salowey, J., Zhou, H., Eronen, P., and H. Tschofenig, + "Transport Layer Security (TLS) Session Resumption without + Server-Side State", RFC 5077, DOI 10.17487/RFC5077, + January 2008, <https://www.rfc-editor.org/info/rfc5077>. + + [RFC5382] Guha, S., Ed., Biswas, K., Ford, B., Sivakumar, S., and P. + Srisuresh, "NAT Behavioral Requirements for TCP", BCP 142, + RFC 5382, DOI 10.17487/RFC5382, October 2008, + <https://www.rfc-editor.org/info/rfc5382>. + + [RFC5905] Mills, D., Martin, J., Ed., Burbank, J., and W. Kasch, + "Network Time Protocol Version 4: Protocol and Algorithms + Specification", RFC 5905, DOI 10.17487/RFC5905, June 2010, + <https://www.rfc-editor.org/info/rfc5905>. + + [RFC6335] Cotton, M., Eggert, L., Touch, J., Westerlund, M., and S. + Cheshire, "Internet Assigned Numbers Authority (IANA) + Procedures for the Management of the Service Name and + Transport Protocol Port Number Registry", BCP 165, + RFC 6335, DOI 10.17487/RFC6335, August 2011, + <https://www.rfc-editor.org/info/rfc6335>. + + [RFC6762] Cheshire, S. and M. Krochmal, "Multicast DNS", RFC 6762, + DOI 10.17487/RFC6762, February 2013, + <https://www.rfc-editor.org/info/rfc6762>. + + [RFC7301] Friedl, S., Popov, A., Langley, A., and E. Stephan, + "Transport Layer Security (TLS) Application-Layer Protocol + Negotiation Extension", RFC 7301, DOI 10.17487/RFC7301, + July 2014, <https://www.rfc-editor.org/info/rfc7301>. + + [RFC7413] Cheng, Y., Chu, J., Radhakrishnan, S., and A. Jain, "TCP + Fast Open", RFC 7413, DOI 10.17487/RFC7413, December 2014, + <https://www.rfc-editor.org/info/rfc7413>. + + [RFC7657] Black, D., Ed. and P. Jones, "Differentiated Services + (Diffserv) and Real-Time Communication", RFC 7657, + DOI 10.17487/RFC7657, November 2015, + <https://www.rfc-editor.org/info/rfc7657>. + + [RFC7838] Nottingham, M., McManus, P., and J. Reschke, "HTTP + Alternative Services", RFC 7838, DOI 10.17487/RFC7838, + April 2016, <https://www.rfc-editor.org/info/rfc7838>. + + [RFC8085] Eggert, L., Fairhurst, G., and G. Shepherd, "UDP Usage + Guidelines", BCP 145, RFC 8085, DOI 10.17487/RFC8085, + March 2017, <https://www.rfc-editor.org/info/rfc8085>. + + [RFC8981] Gont, F., Krishnan, S., Narten, T., and R. Draves, + "Temporary Address Extensions for Stateless Address + Autoconfiguration in IPv6", RFC 8981, + DOI 10.17487/RFC8981, February 2021, + <https://www.rfc-editor.org/info/rfc8981>. + + [RFC9218] Oku, K. and L. Pardue, "Extensible Prioritization Scheme + for HTTP", RFC 9218, DOI 10.17487/RFC9218, June 2022, + <https://www.rfc-editor.org/info/rfc9218>. + + [RFC9221] Pauly, T., Kinnear, E., and D. Schinazi, "An Unreliable + Datagram Extension to QUIC", RFC 9221, + DOI 10.17487/RFC9221, March 2022, + <https://www.rfc-editor.org/info/rfc9221>. + + [SSDP] Donoho, A., Roe, B., Bodlaender, M., Gildred, J., Messer, + A., Kim, Y., Fairman, B., and J. Tourzan, "UPnP Device + Architecture 2.0", 17 April 2020, + <https://openconnectivity.org/upnp-specs/UPnP-arch- + DeviceArchitecture-v2.0-20200417.pdf>. + + [Swett16] Swett, I., "QUIC Deployment Experience @Google", IETF96 + QUIC BoF Presentation, 20 July 2016, + <https://www.ietf.org/proceedings/96/slides/slides-96- + quic-3.pdf>. + + [TAPS-ARCH] + Pauly, T., Trammell, B., Brunstrom, A., Fairhurst, G., and + C. Perkins, "An Architecture for Transport Services", Work + in Progress, Internet-Draft, draft-ietf-taps-arch-14, 27 + September 2022, <https://datatracker.ietf.org/doc/html/ + draft-ietf-taps-arch-14>. + + [TLS13] Rescorla, E., "The Transport Layer Security (TLS) Protocol + Version 1.3", RFC 8446, DOI 10.17487/RFC8446, August 2018, + <https://www.rfc-editor.org/info/rfc8446>. + + [Trammell16] + Trammell, B. and M. Kรผhlewind, "Internet Path Transparency + Measurements using RIPE Atlas", RIPE 72 MAT Presentation, + 25 May 2016, <https://ripe72.ripe.net/wp-content/uploads/ + presentations/86-atlas-udpdiff.pdf>. + +Acknowledgments + + Special thanks to Last Call reviewers Chris Lonvick and Ines Robles. + + This work was partially supported by the European Commission under + Horizon 2020 grant agreement no. 688421 Measurement and Architecture + for a Middleboxed Internet (MAMI) and by the Swiss State Secretariat + for Education, Research, and Innovation under contract no. 15.0268. + This support does not imply endorsement. + +Contributors + + The following people have contributed significant text to or feedback + on this document: + + Gorry Fairhurst + + + Ian Swett + + + Igor Lubashev + + + Lucas Pardue + + + Mike Bishop + + + Mark Nottingham + + + Martin Duke + + + Martin Thomson + + + Sean Turner + + + Tommy Pauly + + +Authors' Addresses + + Mirja Kรผhlewind + Ericsson + Email: mirja.kuehlewind@ericsson.com + + + Brian Trammell + Google Switzerland GmbH + Gustav-Gull-Platz 1 + CH-8004 Zurich + Switzerland + Email: ietf@trammell.ch diff --git a/eval/corpora/rfc/RFC 9312 - Manageability of the QUIC Transport Protocol.txt b/eval/corpora/rfc/RFC 9312 - Manageability of the QUIC Transport Protocol.txt new file mode 100644 index 00000000..a6dff6b8 --- /dev/null +++ b/eval/corpora/rfc/RFC 9312 - Manageability of the QUIC Transport Protocol.txt @@ -0,0 +1,1601 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) M. Kรผhlewind +Request for Comments: 9312 Ericsson +Category: Informational B. Trammell +ISSN: 2070-1721 Google Switzerland GmbH + September 2022 + + + Manageability of the QUIC Transport Protocol + +Abstract + + This document discusses manageability of the QUIC transport protocol + and focuses on the implications of QUIC's design and wire image on + network operations involving QUIC traffic. It is intended as a + "user's manual" for the wire image to provide guidance for network + operators and equipment vendors who rely on the use of transport- + aware network functions. + +Status of This Memo + + This document is not an Internet Standards Track specification; it is + published for informational purposes. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Not all documents + approved by the IESG are candidates for any level of Internet + Standard; see Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9312. + +Copyright Notice + + Copyright (c) 2022 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 2. Features of the QUIC Wire Image + 2.1. QUIC Packet Header Structure + 2.2. Coalesced Packets + 2.3. Use of Port Numbers + 2.4. The QUIC Handshake + 2.5. Integrity Protection of the Wire Image + 2.6. Connection ID and Rebinding + 2.7. Packet Numbers + 2.8. Version Negotiation and Greasing + 3. Network-Visible Information about QUIC Flows + 3.1. Identifying QUIC Traffic + 3.1.1. Identifying Negotiated Version + 3.1.2. First Packet Identification for Garbage Rejection + 3.2. Connection Confirmation + 3.3. Distinguishing Acknowledgment Traffic + 3.4. Server Name Indication (SNI) + 3.4.1. Extracting Server Name Indication (SNI) Information + 3.5. Flow Association + 3.6. Flow Teardown + 3.7. Flow Symmetry Measurement + 3.8. Round-Trip Time (RTT) Measurement + 3.8.1. Measuring Initial RTT + 3.8.2. Using the Spin Bit for Passive RTT Measurement + 4. Specific Network Management Tasks + 4.1. Passive Network Performance Measurement and Troubleshooting + 4.2. Stateful Treatment of QUIC Traffic + 4.3. Address Rewriting to Ensure Routing Stability + 4.4. Server Cooperation with Load Balancers + 4.5. Filtering Behavior + 4.6. UDP Blocking, Throttling, and NAT Binding + 4.7. DDoS Detection and Mitigation + 4.8. Quality of Service Handling and ECMP Routing + 4.9. Handling ICMP Messages + 4.10. Guiding Path MTU + 5. IANA Considerations + 6. Security Considerations + 7. References + 7.1. Normative References + 7.2. Informative References + Acknowledgments + Contributors + Authors' Addresses + +1. Introduction + + QUIC [QUIC-TRANSPORT] is a new transport protocol that is + encapsulated in UDP. QUIC integrates TLS [QUIC-TLS] to encrypt all + payload data and most control information. QUIC version 1 was + designed primarily as a transport for HTTP with the resulting + protocol being known as HTTP/3 [QUIC-HTTP]. + + This document provides guidance for network operations that manage + QUIC traffic. This includes guidance on how to interpret and utilize + information that is exposed by QUIC to the network, requirements and + assumptions of the QUIC design with respect to network treatment, and + a description of how common network management practices will be + impacted by QUIC. + + QUIC is an end-to-end transport protocol; therefore, no information + in the protocol header is intended to be mutable by the network. + This property is enforced through integrity protection of the wire + image [WIRE-IMAGE]. Encryption of most transport-layer control + signaling means that less information is visible to the network in + comparison to TCP. + + Integrity protection can also simplify troubleshooting at the end + points as none of the nodes on the network path can modify transport + layer information. However, it means in-network operations that + depend on modification of data (for examples, see [RFC9065]) are not + possible without the cooperation of a QUIC endpoint. Such + cooperation might be possible with the introduction of a proxy that + authenticates as an endpoint. Proxy operations are not in scope for + this document. + + Network management is not a one-size-fits-all endeavor; for example, + practices considered necessary or even mandatory within enterprise + networks with certain compliance requirements would be impermissible + on other networks without those requirements. Therefore, presence of + a particular practice in this document should not be construed as a + recommendation to apply it. For each practice, this document + describes what is and is not possible with the QUIC transport + protocol as defined. + + This document focuses solely on network management practices that + observe traffic on the wire. For example, replacement of + troubleshooting based on observation with active measurement + techniques is therefore out of scope. A more generalized treatment + of network management operations on encrypted transports is given in + [RFC9065]. + + QUIC-specific terminology used in this document is defined in + [QUIC-TRANSPORT]. + +2. Features of the QUIC Wire Image + + This section discusses aspects of the QUIC transport protocol that + have an impact on the design and operation of devices that forward + QUIC packets. Therefore, this section is primarily considering the + unencrypted part of QUIC's wire image [WIRE-IMAGE], which is defined + as the information available in the packet header in each QUIC + packet, and the dynamics of that information. Since QUIC is a + versioned protocol, the wire image of the header format can also + change from version to version. However, the field that identifies + the QUIC version in some packets and the format of the Version + Negotiation packet are both inspectable and invariant + [QUIC-INVARIANTS]. + + This document addresses version 1 of the QUIC protocol, whose wire + image is fully defined in [QUIC-TRANSPORT] and [QUIC-TLS]. Features + of the wire image described herein may change in future versions of + the protocol except when specified as an invariant [QUIC-INVARIANTS] + and cannot be used to identify QUIC as a protocol or to infer the + behavior of future versions of QUIC. + +2.1. QUIC Packet Header Structure + + QUIC packets may have either a long header or a short header. The + first bit of the QUIC header is the Header Form bit and indicates + which type of header is present. The purpose of this bit is + invariant across QUIC versions. + + The long header exposes more information. It contains a version + number, as well as Source and Destination Connection IDs for + associating packets with a QUIC connection. The definition and + location of these fields in the QUIC long header are invariant for + future versions of QUIC, although future versions of QUIC may provide + additional fields in the long header [QUIC-INVARIANTS]. + + In version 1 of QUIC, the long header is used during connection + establishment to transmit CRYPTO handshake data, perform version + negotiation, retry, and send 0-RTT data. + + Short headers are used after a connection establishment in version 1 + of QUIC and expose only an optional Destination Connection ID and the + initial flags byte with the spin bit for RTT measurement. + + The following information is exposed in QUIC packet headers in all + versions of QUIC (as specified in [QUIC-INVARIANTS]): + + version number: The version number is present in the long header and + identifies the version used for that packet. During Version + Negotiation (see Section 17.2.1 of [QUIC-TRANSPORT] and + Section 2.8), the Version field has a special value (0x00000000) + that identifies the packet as a Version Negotiation packet. QUIC + version 1 uses version 0x00000001. Operators should expect to + observe packets with other version numbers as a result of various + Internet experiments, future standards, and greasing [RFC7801]. + An IANA registry contains the values of all standardized versions + of QUIC, and may contain some proprietary versions (see + Section 22.2 of [QUIC-TRANSPORT]). However, other versions of + QUIC can be expected to be seen in the Internet at any given time. + + Source and Destination Connection ID: Short and long headers carry a + Destination Connection ID, which is a variable-length field. If + the Destination Connection ID is not zero length, it can be used + to identify the connection associated with a QUIC packet for load + balancing and NAT rebinding purposes; see Sections 4.4 and 2.6. + Long packet headers additionally carry a Source Connection ID. + The Source Connection ID is only present on long headers and + indicates the Destination Connection ID that the other endpoint + should use when sending packets. On long header packets, the + length of the connection IDs is also present; on short header + packets, the length of the Destination Connection ID is implicit, + as it is known from preceding long header packets. + + In version 1 of QUIC, the following additional information is + exposed: + + "Fixed Bit": In version 1 of QUIC, the second-most-significant bit + of the first octet is set to 1, unless the packet is a Version + Negotiation packet or an extension is used that modifies the usage + of this bit. If the bit is set to 1, it enables endpoints to + easily demultiplex with other UDP-encapsulated protocols. Even + though this bit is fixed in the version 1 specification, endpoints + might use an extension that varies the bit [QUIC-GREASE]. + Therefore, observers cannot reliably use it as an identifier for + QUIC. + + latency spin bit: The third-most-significant bit of the first octet + in the short header for version 1. The spin bit is set by + endpoints such that tracking edge transitions can be used to + passively observe end-to-end RTT. See Section 3.8.2 for further + details. + + header type: The long header has a 2-bit packet type field following + the Header Form and Fixed Bits. Header types correspond to stages + of the handshake; see Section 17.2 of [QUIC-TRANSPORT] for + details. + + length: The length of the remaining QUIC packet after the Length + field present on long headers. This field is used to implement + coalesced packets during the handshake (see Section 2.2). + + token: Initial packets may contain a token, a variable-length opaque + value optionally sent from client to server, used for validating + the client's address. Retry packets also contain a token, which + can be used by the client in an Initial packet on a subsequent + connection attempt. The length of the token is explicit in both + cases. + + Retry (Section 17.2.5 of [QUIC-TRANSPORT]) and Version Negotiation + (Section 17.2.1 of [QUIC-TRANSPORT]) packets are not encrypted. + Retry packets are integrity protected. Transport parameters are used + to authenticate the contents of Retry packets later in the handshake. + For other kinds of packets, version 1 of QUIC cryptographically + protects other information in the packet headers: + + Packet Number: All packets except Version Negotiation and Retry + packets have an associated packet number; however, this packet + number is encrypted, and therefore not of use to on-path + observers. The offset of the packet number can be decoded in long + headers while it is implicit (depending on Destination Connection + ID length) in short headers. The length of the packet number is + cryptographically protected. + + Key Phase: The Key Phase bit (present in short headers) specifies + the keys used to encrypt the packet to support key rotation. The + Key Phase bit is cryptographically protected. + +2.2. Coalesced Packets + + Multiple QUIC packets may be coalesced into a single UDP datagram + with a datagram carrying one or more long header packets followed by + zero or one short header packets. When packets are coalesced, the + Length fields in the long headers are used to separate QUIC packets; + see Section 12.2 of [QUIC-TRANSPORT]. The Length field is a + variable-length field, and its position in the header also varies + depending on the lengths of the Source and Destination Connection + IDs; see Section 17.2 of [QUIC-TRANSPORT]. + +2.3. Use of Port Numbers + + Applications that have a mapping for TCP and QUIC are expected to use + the same port number for both services. However, as for all other + IETF transports [RFC7605], there is no guarantee that a specific + application will use a given registered port or that a given port + carries traffic belonging to the respective registered service, + especially when application layer information is encrypted. For + example, [QUIC-HTTP] specifies the use of the HTTP Alternative + Services mechanism [RFC7838] for discovery of HTTP/3 services on + other ports. + + Further, as QUIC has a connection ID, it is also possible to maintain + multiple QUIC connections over one 5-tuple (protocol, source, and + destination IP address and source and destination port). However, if + the connection ID is zero length, all packets of the 5-tuple likely + belong to the same QUIC connection. + +2.4. The QUIC Handshake + + New QUIC connections are established using a handshake that is + distinguishable on the wire (see Section 3.1 for details) and + contains some information that can be passively observed. + + To illustrate the information visible in the QUIC wire image during + the handshake, we first show the general communication pattern + visible in the UDP datagrams containing the QUIC handshake. Then, we + examine each of the datagrams in detail. + + The QUIC handshake can normally be recognized on the wire through + four flights of datagrams labeled "Client Initial", "Server Initial", + "Client Completion", and "Server Completion" as illustrated in + Figure 1. + + A handshake starts with the client sending one or more datagrams + containing Initial packets (detailed in Figure 2), which elicits the + Server Initial response (detailed in Figure 3), which typically + contains three types of packets: Initial packet(s) with the beginning + of the server's side of the TLS handshake, Handshake packet(s) with + the rest of the server's portion of the TLS handshake, and 1-RTT + packet(s), if present. + + Client Server + | | + +----Client Initial----------------------->| + +----(zero or more 0-RTT)----------------->| + | | + |<-----------------------Server Initial----+ + |<--------(1-RTT encrypted data starts)----+ + | | + +----Client Completion-------------------->| + +----(1-RTT encrypted data starts)-------->| + | | + |<--------------------Server Completion----+ + | | + + Figure 1: General Communication Pattern Visible in the QUIC Handshake + + As shown here, the client can send 0-RTT data as soon as it has sent + its ClientHello and the server can send 1-RTT data as soon as it has + sent its ServerHello. The Client Completion flight contains at least + one Handshake packet and could also include an Initial packet. + During the handshake, QUIC packets in separate contexts can be + coalesced (see Section 2.2) in order to reduce the number of UDP + datagrams sent during the handshake. + + Handshake packets can arrive out-of-order without impacting the + handshake as long as the reordering was not accompanied by extensive + delays that trigger a spurious Probe Timeout (Section 6.2 of + [QUIC-RECOVERY]). If QUIC packets get lost or reordered, packets + belonging to the same flight might not be observed in close time + succession, though the sequence of the flights will not change + because one flight depends upon the peer's previous flight. + + Datagrams that contain an Initial packet (Client Initial, Server + Initial, and some Client Completion) contain at least 1200 octets of + UDP payload. This protects against amplification attacks and + verifies that the network path meets the requirements for the minimum + QUIC IP packet size; see Section 14 of [QUIC-TRANSPORT]. This is + accomplished by either adding PADDING frames within the Initial + packet, coalescing other packets with the Initial packet, or leaving + unused payload in the UDP packet after the Initial packet. A network + path needs to be able to forward packets of at least this size for + QUIC to be used. + + The content of Initial packets is encrypted using Initial Secrets, + which are derived from a per-version constant and the client's + Destination Connection ID. That content is therefore observable by + any on-path device that knows the per-version constant and is + considered visible in this illustration. The content of QUIC + Handshake packets is encrypted using keys established during the + initial handshake exchange and is therefore not visible. + + Initial, Handshake, and 1-RTT packets belong to different + cryptographic and transport contexts. The Client Completion + (Figure 4) and the Server Completion (Figure 5) flights conclude the + Initial and Handshake contexts by sending final acknowledgments and + CRYPTO frames. + + +----------------------------------------------------------+ + | UDP header (source and destination UDP ports) | + +----------------------------------------------------------+ + | QUIC long header (type = Initial, Version, DCID, SCID) (Length) + +----------------------------------------------------------+ | + | QUIC CRYPTO frame header | | + +----------------------------------------------------------+ | + | | TLS ClientHello (incl. TLS SNI) | | | + +----------------------------------------------------------+ | + | QUIC PADDING frames | | + +----------------------------------------------------------+<-+ + + Figure 2: Example Client Initial Datagram Without 0-RTT + + A Client Initial packet exposes the Version, Source, and Destination + Connection IDs without encryption. The payload of the Initial packet + is protected using the Initial secret. The complete TLS ClientHello, + including any TLS Server Name Indication (SNI) present, is sent in + one or more CRYPTO frames across one or more QUIC Initial packets. + + +------------------------------------------------------------+ + | UDP header (source and destination UDP ports) | + +------------------------------------------------------------+ + | QUIC long header (type = Initial, Version, DCID, SCID) (Length) + +------------------------------------------------------------+ | + | QUIC CRYPTO frame header | | + +------------------------------------------------------------+ | + | TLS ServerHello | | + +------------------------------------------------------------+ | + | QUIC ACK frame (acknowledging client hello) | | + +------------------------------------------------------------+<-+ + | QUIC long header (type = Handshake, Version, DCID, SCID) (Length) + +------------------------------------------------------------+ | + | encrypted payload (presumably CRYPTO frames) | | + +------------------------------------------------------------+<-+ + | QUIC short header | + +------------------------------------------------------------+ + | 1-RTT encrypted payload | + +------------------------------------------------------------+ + + Figure 3: Coalesced Server Initial Datagram Pattern + + The Server Initial datagram also exposes the version number and the + Source and Destination Connection IDs in the clear; the payload of + the Initial packet is protected using the Initial secret. + + +------------------------------------------------------------+ + | UDP header (source and destination UDP ports) | + +------------------------------------------------------------+ + | QUIC long header (type = Initial, Version, DCID, SCID) (Length) + +------------------------------------------------------------+ | + | QUIC ACK frame (acknowledging Server Initial) | | + +------------------------------------------------------------+<-+ + | QUIC long header (type = Handshake, Version, DCID, SCID) (Length) + +------------------------------------------------------------+ | + | encrypted payload (presumably CRYPTO/ACK frames) | | + +------------------------------------------------------------+<-+ + | QUIC short header | + +------------------------------------------------------------+ + | 1-RTT encrypted payload | + +------------------------------------------------------------+ + + Figure 4: Coalesced Client Completion Datagram Pattern + + The Client Completion flight does not expose any additional + information; however, as the Destination Connection ID is server- + selected, it usually is not the same ID that is sent in the Client + Initial. Client Completion flights contain 1-RTT packets that + indicate the handshake has completed (see Section 3.2) on the client + and for three-way handshake RTT estimation as in Section 3.8. + + +------------------------------------------------------------+ + | UDP header (source and destination UDP ports) | + +------------------------------------------------------------+ + | QUIC long header (type = Handshake, Version, DCID, SCID) (Length) + +------------------------------------------------------------+ | + | encrypted payload (presumably ACK frame) | | + +------------------------------------------------------------+<-+ + | QUIC short header | + +------------------------------------------------------------+ + | 1-RTT encrypted payload | + +------------------------------------------------------------+ + + Figure 5: Coalesced Server Completion Datagram Pattern + + Similar to Client Completion, Server Completion does not expose + additional information; observing it serves only to determine that + the handshake has completed. + + When the client uses 0-RTT data, the Client Initial flight can also + include one or more 0-RTT packets as shown in Figure 6. + + +----------------------------------------------------------+ + | UDP header (source and destination UDP ports) | + +----------------------------------------------------------+ + | QUIC long header (type = Initial, Version, DCID, SCID) (Length) + +----------------------------------------------------------+ | + | QUIC CRYPTO frame header | | + +----------------------------------------------------------+ | + | TLS ClientHello (incl. TLS SNI) | | + +----------------------------------------------------------+<-+ + | QUIC long header (type = 0-RTT, Version, DCID, SCID) (Length) + +----------------------------------------------------------+ | + | 0-RTT encrypted payload | | + +----------------------------------------------------------+<-+ + + Figure 6: Coalesced 0-RTT Client Initial Datagram + + When a 0-RTT packet is coalesced with an Initial packet, the datagram + will be padded to 1200 bytes. Additional datagrams containing only + 0-RTT packets with long headers can be sent after the client Initial + packet, which contains more 0-RTT data. The amount of 0-RTT + protected data that can be sent in the first flight is limited by the + initial congestion window, typically to around 10 packets (see + Section 7.2 of [QUIC-RECOVERY]). + +2.5. Integrity Protection of the Wire Image + + As soon as the cryptographic context is established, all information + in the QUIC header, including exposed information, is integrity + protected. Further, information that was exposed in packets sent + before the cryptographic context was established is validated during + the cryptographic handshake. Therefore, devices on path cannot alter + any information or bits in QUIC packets. Such alterations would + cause the integrity check to fail, which results in the receiver + discarding the packet. Some parts of Initial packets could be + altered by removing and reapplying the authenticated encryption + without immediate discard at the receiver. However, the + cryptographic handshake validates most fields and any modifications + in those fields will result in a connection establishment failure + later. + +2.6. Connection ID and Rebinding + + The connection ID in the QUIC packet headers allows association of + QUIC packets using information independent of the 5-tuple. This + allows rebinding of a connection after one of the endpoints (usually + the client) has experienced an address change. Further, it can be + used by in-network devices to ensure that related 5-tuple flows are + appropriately balanced together (see Section 4.4). + + Client and server each choose a connection ID during the handshake; + for example, a server might request that a client use a connection + ID, whereas the client might choose a zero-length value. Connection + IDs for either endpoint may change during the lifetime of a + connection, with the new connection ID being supplied via encrypted + frames (see Section 5.1 of [QUIC-TRANSPORT]). Therefore, observing a + new connection ID does not necessarily indicate a new connection. + + [QUIC-LB] specifies algorithms for encoding the server mapping in a + connection ID in order to share this information with selected on- + path devices such as load balancers. Server mappings should only be + exposed to selected entities. Uncontrolled exposure would allow + linkage of multiple IP addresses to the same host if the server also + supports migration that opens an attack vector on specific servers or + pools. The best way to obscure an encoding is to appear random to + any other observers, which is most rigorously achieved with + encryption. As a result, any attempt to infer information from + specific parts of a connection ID is unlikely to be useful. + +2.7. Packet Numbers + + The Packet Number field is always present in the QUIC packet header + in version 1; however, it is always encrypted. The encryption key + for packet number protection on Initial packets (which are sent + before cryptographic context establishment) is specific to the QUIC + version while packet number protection on subsequent packets uses + secrets derived from the end-to-end cryptographic context. Packet + numbers are therefore not part of the wire image that is visible to + on-path observers. + +2.8. Version Negotiation and Greasing + + Version Negotiation packets are used by the server to indicate that a + requested version from the client is not supported (see Section 6 of + [QUIC-TRANSPORT]). Version Negotiation packets are not intrinsically + protected, but future QUIC versions could use later encrypted + messages to verify that they were authentic. Therefore, any + modification of this list will be detected and may cause the + endpoints to terminate the connection attempt. + + Also note that the list of versions in the Version Negotiation packet + may contain reserved versions. This mechanism is used to avoid + ossification in the implementation of the selection mechanism. + Further, a client may send an Initial packet with a reserved version + number to trigger version negotiation. In the Version Negotiation + packet, the connection IDs of the client's Initial packet are + reflected to provide a proof of return-routability. Therefore, + changing this information will also cause the connection to fail. + + QUIC is expected to evolve rapidly. Therefore, new versions (both + experimental and IETF standard versions) will be deployed on the + Internet more often than with other commonly deployed Internet and + transport-layer protocols. Use of the Version field for traffic + recognition will therefore behave differently than with these + protocols. Using a particular version number to recognize valid QUIC + traffic is likely to persistently miss a fraction of QUIC flows and + completely fail in the near future. Reliance on the Version field + for the purpose of admission control is also likely to lead to + unintended failure modes. Admission of QUIC traffic regardless of + version avoids these failure modes, avoids unnecessary deployment + delays, and supports continuous version-based evolution. + +3. Network-Visible Information about QUIC Flows + + This section addresses the different kinds of observations and + inferences that can be made about QUIC flows by a passive observer in + the network based on the wire image in Section 2. Here, we assume a + bidirectional observer (one that can see packets in both directions + in the sequence in which they are carried on the wire) unless noted, + but typically without access to any keying information. + +3.1. Identifying QUIC Traffic + + The QUIC wire image is not specifically designed to be + distinguishable from other UDP traffic by a passive observer in the + network. While certain QUIC applications may be heuristically + identifiable on a per-application basis, there is no general method + for distinguishing QUIC traffic from otherwise unclassifiable UDP + traffic on a given link. Therefore, any unrecognized UDP traffic may + be QUIC traffic. + + At the time of writing, two application bindings for QUIC have been + published or adopted by the IETF: HTTP/3 [QUIC-HTTP] and DNS over + Dedicated QUIC Connections [RFC9250]. These are both known to have + active Internet deployments, so an assumption that all QUIC traffic + is HTTP/3 is not valid. HTTP/3 uses UDP port 443 by convention but + various methods can be used to specify alternate port numbers. Other + applications (e.g., Microsoft's SMB over QUIC) also use UDP port 443 + by default. Therefore, simple assumptions about whether a given flow + is using QUIC (or indeed which application might be using QUIC) based + solely upon a UDP port number may not hold; see Section 5 of + [RFC7605]. + + While the second-most-significant bit (0x40) of the first octet is + set to 1 in most QUIC packets of the current version (see Section 2.1 + and Section 17 of [QUIC-TRANSPORT]), this method of recognizing QUIC + traffic is not reliable. First, it only provides one bit of + information and is prone to collision with UDP-based protocols other + than those considered in [RFC7983]. Second, this feature of the wire + image is not invariant [QUIC-INVARIANTS] and may change in future + versions of the protocol or even be negotiated during the handshake + via the use of an extension [QUIC-GREASE]. + + Even though transport parameters transmitted in the client's Initial + packet are observable by the network, they cannot be modified by the + network without causing a connection failure. Further, the reply + from the server cannot be observed, so observers on the network + cannot know which parameters are actually in use. + +3.1.1. Identifying Negotiated Version + + An in-network observer assuming that a set of packets belongs to a + QUIC flow might infer the version number in use by observing the + handshake. If the version number in an Initial packet of the server + response is subsequently seen in a packet from the client, that + version has been accepted by both endpoints to be used for the rest + of the connection (see Section 2 of [QUIC-VERSION-NEGOTIATION]). + + The negotiated version cannot be identified for flows in which a + handshake is not observed, such as in the case of connection + migration. However, it might be possible to associate a flow with a + flow for which a version has been identified; see Section 3.5. + +3.1.2. First Packet Identification for Garbage Rejection + + A related question is whether the first packet of a given flow on a + port known to be associated with QUIC is a valid QUIC packet. This + determination supports in-network filtering of garbage UDP packets + (reflection attacks, random backscatter, etc.). While heuristics + based on the first byte of the packet (packet type) could be used to + separate valid from invalid first packet types, the deployment of + such heuristics is not recommended as bits in the first byte may have + different meanings in future versions of the protocol. + +3.2. Connection Confirmation + + This document focuses on QUIC version 1, and this Connection + Confirmation section applies only to packets belonging to QUIC + version 1 flows; for purposes of on-path observation, it assumes that + these packets have been identified as such through the observation of + a version number exchange as described above. + + Connection establishment uses Initial and Handshake packets + containing a TLS handshake and Retry packets that do not contain + parts of the handshake. Connection establishment can therefore be + detected using heuristics similar to those used to detect TLS over + TCP. A client initiating a connection may also send data in 0-RTT + packets directly after the Initial packet containing the TLS + ClientHello. Since packets may be reordered or lost in the network, + 0-RTT packets could be seen before the Initial packet. + + Note that in this version of QUIC, clients send Initial packets + before servers do, servers send Handshake packets before clients do, + and only clients send Initial packets with tokens. Therefore, an + endpoint can be identified as a client or server by an on-path + observer. An attempted connection after Retry can be detected by + correlating the contents of the Retry packet with the Token and the + Destination Connection ID fields of the new Initial packet. + +3.3. Distinguishing Acknowledgment Traffic + + Some deployed in-network functions distinguish packets that carry + only acknowledgment (ACK-only) information from packets carrying + upper-layer data in order to attempt to enhance performance (for + example, by queuing ACKs differently or manipulating ACK signaling + [RFC3449]). Distinguishing ACK packets is possible in TCP, but is + not supported by QUIC since acknowledgment signaling is carried + inside QUIC's encrypted payload and ACK manipulation is impossible. + Specifically, heuristics attempting to distinguish ACK-only packets + from payload-carrying packets based on packet size are likely to fail + and are not recommended to use as a way to construe internals of + QUIC's operation as those mechanisms can change, e.g., due to the use + of extensions. + +3.4. Server Name Indication (SNI) + + The client's TLS ClientHello may contain a Server Name Indication + (SNI) extension [RFC6066] by which the client reveals the name of the + server it intends to connect to in order to allow the server to + present a certificate based on that name. If present, SNI + information is available to unidirectional observers on the client- + to-server path if it. + + The TLS ClientHello may also contain an Application-Layer Protocol + Negotiation (ALPN) extension [RFC7301], by which the client exposes + the names of application-layer protocols it supports; an observer can + deduce that one of those protocols will be used if the connection + continues. + + Work is currently underway in the TLS working group to encrypt the + contents of the ClientHello in TLS 1.3 [TLS-ECH]. This would make + SNI-based application identification impossible by on-path + observation for QUIC and other protocols that use TLS. + +3.4.1. Extracting Server Name Indication (SNI) Information + + If the ClientHello is not encrypted, SNI can be derived from the + client's Initial packets by calculating the Initial secret to decrypt + the packet payload and parsing the QUIC CRYPTO frames containing the + TLS ClientHello. + + As both the derivation of the Initial secret and the structure of the + Initial packet itself are version specific, the first step is always + to parse the version number (the second through fifth bytes of the + long header). Note that only long header packets carry the version + number, so it is necessary to also check if the first bit of the QUIC + packet is set to 1, which indicates a long header. + + Note that proprietary QUIC versions that have been deployed before + standardization might not set the first bit in a QUIC long header + packet to 1. However, it is expected that these versions will + gradually disappear over time and therefore do not require any + special consideration or treatment. + + When the version has been identified as QUIC version 1, the packet + type needs to be verified as an Initial packet by checking that the + third and fourth bits of the header are both set to 0. Then, the + Destination Connection ID needs to be extracted from the packet. The + Initial secret is calculated using the version-specific Initial salt + as described in Section 5.2 of [QUIC-TLS]. The length of the + connection ID is indicated in the 6th byte of the header followed by + the connection ID itself. + + Note that subsequent Initial packets might contain a Destination + Connection ID other than the one used to generate the Initial secret. + Therefore, attempts to decrypt these packets using the procedure + above might fail unless the Initial secret is retained by the + observer. + + To determine the end of the packet header and find the start of the + payload, the Packet Number Length, the Source Connection ID Length, + and the Token Length need to be extracted. The Packet Number Length + is defined by the seventh and eighth bits of the header as described + in Section 17.2 of [QUIC-TRANSPORT], but is protected as described in + Section 5.4 of [QUIC-TLS]. The Source Connection ID Length is + specified in the byte after the Destination Connection ID. The Token + Length, which follows the Source Connection ID, is a variable-length + integer as specified in Section 16 of [QUIC-TRANSPORT]. + + After decryption, the client's Initial packets can be parsed to + detect the CRYPTO frames that contain the TLS ClientHello, which then + can be parsed similarly to TLS over TCP connections. Note that there + can be multiple CRYPTO frames spread out over one or more Initial + packets and they might not be in order, so reassembling the CRYPTO + stream by parsing offsets and lengths is required. Further, the + client's Initial packets may contain other frames, so the first bytes + of each frame need to be checked to identify the frame type and + determine whether the frame can be skipped over. Note that the + length of the frames is dependent on the frame type; see Section 18 + of [QUIC-TRANSPORT]. For example, PADDING frames (each consisting of + a single zero byte) may occur before, after, or between CRYPTO + frames. However, extensions might define additional frame types. If + an unknown frame type is encountered, it is impossible to know the + length of that frame, which prevents skipping over it; therefore, + parsing fails. + +3.5. Flow Association + + The QUIC connection ID (see Section 2.6) is designed to allow a + coordinating on-path device, such as a load balancer, to associate + two flows when one of the endpoints changes address. This change can + be due to NAT rebinding or address migration. + + The connection ID must change upon intentional address change by an + endpoint and connection ID negotiation is encrypted; therefore, it is + not possible for a passive observer to link intended changes of + address using the connection ID. + + When one endpoint's address unintentionally changes, as is the case + with NAT rebinding, an on-path observer may be able to use the + connection ID to associate the flow on the new address with the flow + on the old address. + + A network function that attempts to use the connection ID to + associate flows must be robust to the failure of this technique. + Since the connection ID may change multiple times during the lifetime + of a connection, packets with the same 5-tuple but different + connection IDs might or might not belong to the same connection. + Likewise, packets with the same connection ID but different 5-tuples + might not belong to the same connection either. + + Connection IDs should be treated as opaque; see Section 4.4 for + caveats regarding connection ID selection at servers. + +3.6. Flow Teardown + + QUIC does not expose the end of a connection; the only indication to + on-path devices that a flow has ended is that packets are no longer + observed. Therefore, stateful devices on path such as NATs and + firewalls must use idle timeouts to determine when to drop state for + QUIC flows; see Section 4.2. + +3.7. Flow Symmetry Measurement + + QUIC explicitly exposes which side of a connection is a client and + which side is a server during the handshake. In addition, the + symmetry of a flow (whether it is primarily client-to-server, + primarily server-to-client, or roughly bidirectional, as input to + basic traffic classification techniques) can be inferred through the + measurement of data rate in each direction. Note that QUIC packets + containing only control frames (such as ACK-only packets) may be + padded. Padding, though optional, may conceal connection roles or + flow symmetry information. + +3.8. Round-Trip Time (RTT) Measurement + + The round-trip time (RTT) of QUIC flows can be inferred by + observation once per flow during the handshake in passive TCP + measurement; this requires parsing of the QUIC packet header and + recognition of the handshake, as illustrated in Section 2.4. It can + also be inferred during the flow's lifetime if the endpoints use the + spin bit facility described below and in Section 17.3.1 of + [QUIC-TRANSPORT]. RTT measurement is available to unidirectional + observers when the spin bit is enabled. + +3.8.1. Measuring Initial RTT + + In the common case, the delay between the client's Initial packet + (containing the TLS ClientHello) and the server's Initial packet + (containing the TLS ServerHello) represents the RTT component on the + path between the observer and the server. The delay between the + server's first Handshake packet and the Handshake packet sent by the + client represents the RTT component on the path between the observer + and the client. While the client may send 0-RTT packets after the + Initial packet during connection re-establishment, these can be + ignored for RTT measurement purposes. + + Handshake RTT can be measured by adding the client-to-observer and + observer-to-server RTT components together. This measurement + necessarily includes all transport- and application-layer delay at + both endpoints. + +3.8.2. Using the Spin Bit for Passive RTT Measurement + + The spin bit provides a version-specific method to measure per-flow + RTT from observation points on the network path throughout the + duration of a connection. See Section 17.4 of [QUIC-TRANSPORT] for + the definition of the spin bit in Version 1 of QUIC. Endpoint + participation in spin bit signaling is optional. While its location + is fixed in this version of QUIC, an endpoint can unilaterally choose + to not support "spinning" the bit. + + Use of the spin bit for RTT measurement by devices on path is only + possible when both endpoints enable it. Some endpoints may disable + use of the spin bit by default, others only in specific deployment + scenarios, e.g., for servers and clients where the RTT would reveal + the presence of a VPN or proxy. To avoid making these connections + identifiable based on the usage of the spin bit, all endpoints + randomly disable "spinning" for at least one eighth of connections, + even if otherwise enabled by default. An endpoint not participating + in spin bit signaling for a given connection can use a fixed spin + value for the duration of the connection or can set the bit randomly + on each packet sent. + + When in use, the latency spin bit in each direction changes value + once per RTT any time that both endpoints are sending packets + continuously. An on-path observer can observe the time difference + between edges (changes from 1 to 0 or 0 to 1) in the spin bit signal + in a single direction to measure one sample of end-to-end RTT. This + mechanism follows the principles of protocol measurability laid out + in [IPIM]. + + Note that this measurement, as with passive RTT measurement for TCP, + includes all transport protocol delay (e.g., delayed sending of + acknowledgments) and/or application layer delay (e.g., waiting for a + response to be generated). It therefore provides devices on path a + good instantaneous estimate of the RTT as experienced by the + application. + + However, application-limited and flow-control-limited senders can + have application- and transport-layer delay, respectively, that are + much greater than network RTT. For example, if the sender only sends + small amounts of application traffic periodically, where the + periodicity is longer than the RTT, spin bit measurements provide + information about the application period rather than network RTT. + + Since the spin bit logic at each endpoint considers only samples from + packets that advance the largest packet number, signal generation + itself is resistant to reordering. However, reordering can cause + problems at an observer by causing spurious edge detection and + therefore inaccurate (i.e., lower) RTT estimates, if reordering + occurs across a spin bit flip in the stream. + + Simple heuristics based on the observed data rate per flow or changes + in the RTT series can be used to reject bad RTT samples due to lost + or reordered edges in the spin signal, as well as application or flow + control limitation; for example, QoF [TMA-QOF] rejects component RTTs + significantly higher than RTTs over the history of the flow. These + heuristics may use the handshake RTT as an initial RTT estimate for a + given flow. Usually such heuristics would also detect if the spin is + either constant or randomly set for a connection. + + An on-path observer that can see traffic in both directions (from + client to server and from server to client) can also use the spin bit + to measure "upstream" and "downstream" component RTT; i.e, the + component of the end-to-end RTT attributable to the paths between the + observer and the server and between the observer and the client, + respectively. It does this by measuring the delay between a spin + edge observed in the upstream direction and that observed in the + downstream direction, and vice versa. + + Raw RTT samples generated using these techniques can be processed in + various ways to generate useful network performance metrics. A + simple linear smoothing or moving minimum filter can be applied to + the stream of RTT samples to get a more stable estimate of + application-experienced RTT. RTT samples measured from the spin bit + can also be used to generate RTT distribution information, including + minimum RTT (which approximates network RTT over longer time windows) + and RTT variance (which approximates one-way packet delay variance as + seen by an application end-point). + +4. Specific Network Management Tasks + + In this section, we review specific network management and + measurement techniques and how QUIC's design impacts them. + +4.1. Passive Network Performance Measurement and Troubleshooting + + Limited RTT measurement is possible by passive observation of QUIC + traffic; see Section 3.8. No passive measurement of loss is possible + with the present wire image. Limited observation of upstream + congestion may be possible via the observation of Congestion + Experienced (CE) markings in the IP header [RFC3168] on ECN-enabled + QUIC traffic. + + On-path devices can also make measurements of RTT, loss, and other + performance metrics when information is carried in an additional + network-layer packet header (Section 6 of [RFC9065] describes the use + of Operations, Administration, and Management (OAM) information). + Using network-layer approaches also has the advantage that common + observation and analysis tools can be consistently used for multiple + transport protocols; however, these techniques are often limited to + measurements within one or multiple cooperating domains. + +4.2. Stateful Treatment of QUIC Traffic + + Stateful treatment of QUIC traffic (e.g., at a firewall or NAT + middlebox) is possible through QUIC traffic and version + identification (Section 3.1) and observation of the handshake for + connection confirmation (Section 3.2). The lack of any visible end- + of-flow signal (Section 3.6) means that this state must be purged + either through timers or least-recently-used eviction depending on + application requirements. + + While QUIC has no clear network-visible end-of-flow signal and + therefore does require timer-based state removal, the QUIC handshake + indicates confirmation by both ends of a valid bidirectional + transmission. As soon as the handshake completed, timers should be + set long enough to also allow for short idle time during a valid + transmission. + + [RFC4787] requires a network state timeout that is not less than 2 + minutes for most UDP traffic. However, in practice, a QUIC endpoint + can experience lower timeouts in the range of 30 to 60 seconds + [QUIC-TIMEOUT]. + + In contrast, [RFC5382] recommends a state timeout of more than 2 + hours for TCP given that TCP is a connection-oriented protocol with + well-defined closure semantics. Even though QUIC has explicitly been + designed to tolerate NAT rebindings, decreasing the NAT timeout is + not recommended as it may negatively impact application performance + or incentivize endpoints to send very frequent keep-alive packets. + + Therefore, a state timeout of at least two minutes is recommended for + QUIC traffic, even when lower state timeouts are used for other UDP + traffic. + + If state is removed too early, this could lead to black-holing of + incoming packets after a short idle period. To detect this + situation, a timer at the client needs to expire before a re- + establishment can happen (if at all), which would lead to + unnecessarily long delays in an otherwise working connection. + + Furthermore, not all endpoints use routing architectures where + connections will survive a port or address change. Even when the + client revives the connection, a NAT rebinding can cause a routing + mismatch where a packet is not even delivered to the server that + might support address migration. For these reasons, the limits in + [RFC4787] are important to avoid black-holing of packets (and hence + avoid interrupting the flow of data to the client), especially where + devices are able to distinguish QUIC traffic from other UDP payloads. + + The QUIC header optionally contains a connection ID, which could + provide additional entropy beyond the 5-tuple. The QUIC handshake + needs to be observed in order to understand whether the connection ID + is present and what length it has. However, connection IDs may be + renegotiated after the handshake, and this renegotiation is not + visible to the path. Therefore, using the connection ID as a flow + key field for stateful treatment of flows is not recommended as + connection ID changes will cause undetectable and unrecoverable loss + of state in the middle of a connection. In particular, the use of + the connection ID for functions that require state to make a + forwarding decision is not viable as it will break connectivity, or + at minimum, cause long timeout-based delays before this problem is + detected by the endpoints and the connection can potentially be re- + established. + + Use of connection IDs is specifically discouraged for NAT + applications. If a NAT hits an operational limit, it is recommended + to rather drop the initial packets of a flow (see also Section 4.5), + which potentially triggers TCP fallback. Use of the connection ID to + multiplex multiple connections on the same IP address/port pair is + not a viable solution as it risks connectivity breakage in case the + connection ID changes. + +4.3. Address Rewriting to Ensure Routing Stability + + While QUIC's migration capability makes it possible for a connection + to survive client address changes, this does not work if the routers + or switches in the server infrastructure route using the address-port + 4-tuple. If infrastructure routes on addresses only, NAT rebinding + or address migration will cause packets to be delivered to the wrong + server. [QUIC-LB] describes a way to addresses this problem by + coordinating the selection and use of connection IDs between load + balancers and servers. + + Applying address translation at a middlebox to maintain a stable + address-port mapping for flows based on connection ID might seem like + a solution to this problem. However, hiding information about the + change of the IP address or port conceals important and security- + relevant information from QUIC endpoints, and as such, would + facilitate amplification attacks (see Section 8 of [QUIC-TRANSPORT]). + A NAT function that hides peer address changes prevents the other end + from detecting and mitigating attacks as the endpoint cannot verify + connectivity to the new address using QUIC PATH_CHALLENGE and + PATH_RESPONSE frames. + + In addition, a change of IP address or port is also an input signal + to other internal mechanisms in QUIC. When a path change is + detected, path-dependent variables like congestion control parameters + will be reset, which protects the new path from overload. + +4.4. Server Cooperation with Load Balancers + + In the case of networking architectures that include load balancers, + the connection ID can be used as a way for the server to signal + information about the desired treatment of a flow to the load + balancers. Guidance on assigning connection IDs is given in + [QUIC-APPLICABILITY]. [QUIC-LB] describes a system for coordinating + selection and use of connection IDs between load balancers and + servers. + +4.5. Filtering Behavior + + [RFC4787] describes possible packet-filtering behaviors that relate + to NATs but are often also used in other scenarios where packet + filtering is desired. Though the guidance there holds, a + particularly unwise behavior admits a handful of UDP packets and then + makes a decision to whether or not filter later packets in the same + connection. QUIC applications are encouraged to fall back to TCP if + early packets do not arrive at their destination + [QUIC-APPLICABILITY], as QUIC is based on UDP and there are known + blocks of UDP traffic (see Section 4.6). Admitting a few packets + allows the QUIC endpoint to determine that the path accepts QUIC. + Sudden drops afterwards will result in slow and costly timeouts + before abandoning the connection. + +4.6. UDP Blocking, Throttling, and NAT Binding + + Today, UDP is the most prevalent DDoS vector, since it is easy for + compromised non-admin applications to send a flood of large UDP + packets (while with TCP the attacker gets throttled by the congestion + controller) or to craft reflection and amplification attacks; + therefore, some networks block UDP traffic. With increased + deployment of QUIC, there is also an increased need to allow UDP + traffic on ports used for QUIC. However, if UDP is generally enabled + on these ports, UDP flood attacks may also use the same ports. One + possible response to this threat is to throttle UDP traffic on the + network, allocating a fixed portion of the network capacity to UDP + and blocking UDP datagrams over that cap. As the portion of QUIC + traffic compared to TCP is also expected to increase over time, using + such a limit is not recommended; if this is done, limits might need + to be adapted dynamically. + + Further, if UDP traffic is desired to be throttled, it is recommended + to block individual QUIC flows entirely rather than dropping packets + indiscriminately. When the handshake is blocked, QUIC-capable + applications may fall back to TCP. However, blocking a random + fraction of QUIC packets across 4-tuples will allow many QUIC + handshakes to complete, preventing TCP fallback, but these + connections will suffer from severe packet loss (see also + Section 4.5). Therefore, UDP throttling should be realized by per- + flow policing as opposed to per-packet policing. Note that this per- + flow policing should be stateless to avoid problems with stateful + treatment of QUIC flows (see Section 4.2), for example, blocking a + portion of the space of values of a hash function over the addresses + and ports in the UDP datagram. While QUIC endpoints are often able + to survive address changes, e.g., by NAT rebindings, blocking a + portion of the traffic based on 5-tuple hashing increases the risk of + black-holing an active connection when the address changes. + + Note that some source ports are assumed to be reflection attack + vectors by some servers; see Section 8.1 of [QUIC-APPLICABILITY]. As + a result, NAT binding to these source ports can result in that + traffic being blocked. + +4.7. DDoS Detection and Mitigation + + On-path observation of the transport headers of packets can be used + for various security functions. For example, Denial of Service (DoS) + and Distributed DoS (DDoS) attacks against the infrastructure or + against an endpoint can be detected and mitigated by characterizing + anomalous traffic. Other uses include support for security audits + (e.g., verifying the compliance with cipher suites), client and + application fingerprinting for inventory, and providing alerts for + network intrusion detection and other next-generation firewall + functions. + + Current practices in detection and mitigation of DDoS attacks + generally involve classification of incoming traffic (as packets, + flows, or some other aggregate) into "good" (productive) and "bad" + (DDoS) traffic, and then differential treatment of this traffic to + forward only good traffic. This operation is often done in a + separate specialized mitigation environment through which all traffic + is filtered; a generalized architecture for separation of concerns in + mitigation is given in [DOTS-ARCH]. + + Efficient classification of this DDoS traffic in the mitigation + environment is key to the success of this approach. Limited first + packet garbage detection as in Section 3.1.2 and stateful tracking of + QUIC traffic as mentioned in Section 4.2 above may be useful during + classification. + + Note that using a connection ID to support connection migration + renders 5-tuple-based filtering insufficient to detect active flows + and requires more state to be maintained by DDoS defense systems if + support of migration of QUIC flows is desired. For the common case + of NAT rebinding, where the client's address changes without the + client's intent or knowledge, DDoS defense systems can detect a + change in the client's endpoint address by linking flows based on the + server's connection IDs. However, QUIC's linkability resistance + ensures that a deliberate connection migration is accompanied by a + change in the connection ID. In this case, the connection ID cannot + be used to distinguish valid, active traffic from new attack traffic. + + It is also possible for endpoints to directly support security + functions such as DoS classification and mitigation. Endpoints can + cooperate with an in-network device directly by e.g., sharing + information about connection IDs. + + Another potential method could use an on-path network device that + relies on pattern inferences in the traffic and heuristics or machine + learning instead of processing observed header information. + + However, it is questionable whether connection migrations must be + supported during a DDoS attack. While unintended migration without a + connection ID change can be supported much easier, it might be + acceptable to not support migrations of active QUIC connections that + are not visible to the network functions performing the DDoS + detection. As soon as the connection blocking is detected by the + client, the client may be able to rely on the 0-RTT data mechanism + provided by QUIC. When clients migrate to a new path, they should be + prepared for the migration to fail and attempt to reconnect quickly. + + Beyond in-network DDoS protection mechanisms, TCP SYN cookies + [RFC4987] are a well-established method of mitigating some kinds of + TCP DDoS attacks. QUIC Retry packets are the functional analogue to + SYN cookies, forcing clients to prove possession of their IP address + before committing server state. However, there are safeguards in + QUIC against unsolicited injection of these packets by intermediaries + who do not have consent of the end server. See [QUIC-RETRY] for + standard ways for intermediaries to send Retry packets on behalf of + consenting servers. + +4.8. Quality of Service Handling and ECMP Routing + + It is expected that any QoS handling in the network, e.g., based on + use of Diffserv Code Points (DSCPs) [RFC2475] as well as Equal-Cost + Multi-Path (ECMP) routing, is applied on a per-flow basis (and not + per-packet) and as such that all packets belonging to the same active + QUIC connection get uniform treatment. + + Using ECMP to distribute packets from a single flow across multiple + network paths or any other nonuniform treatment of packets belong to + the same connection could result in variations in order, delivery + rate, and drop rate. As feedback about loss or delay of each packet + is used as input to the congestion controller, these variations could + adversely affect performance. Depending on the loss recovery + mechanism that is implemented, QUIC may be more tolerant of packet + reordering than typical TCP traffic (see Section 2.7). However, the + recovery mechanism used by a flow cannot be known by the network and + therefore reordering tolerance should be considered as unknown. + + Note that the 5-tuple of a QUIC connection can change due to + migration. In this case different flows are observed by the path and + may be treated differently, as congestion control is usually reset on + migration (see also Section 3.5). + +4.9. Handling ICMP Messages + + Datagram Packetization Layer PMTU Discovery (DPLPMTUD) can be used by + QUIC to probe for the supported PMTU. DPLPMTUD optionally uses ICMP + messages (e.g., IPv6 Packet Too Big (PTB) messages). Given known + attacks with the use of ICMP messages, the use of DPLPMTUD in QUIC + has been designed to safely use but not rely on receiving ICMP + feedback (see Section 14.2.1 of [QUIC-TRANSPORT]). + + Networks are recommended to forward these ICMP messages and retain as + much of the original packet as possible without exceeding the minimum + MTU for the IP version when generating ICMP messages as recommended + in [RFC1812] and [RFC4443]. + +4.10. Guiding Path MTU + + Some network segments support 1500-byte packets, but can only do so + by fragmenting at a lower layer before traversing a network segment + with a smaller MTU, and then reassembling within the network segment. + This is permissible even when the IP layer is IPv6 or IPv4 with the + Don't Fragment (DF) bit set, because fragmentation occurs below the + IP layer. However, this process can add to compute and memory costs, + leading to a bottleneck that limits network capacity. In such + networks, this generates a desire to influence a majority of senders + to use smaller packets to avoid exceeding limited reassembly + capacity. + + For TCP, Maximum Segment Size (MSS) clamping (Section 3.2 of + [RFC4459]) is often used to change the sender's TCP maximum segment + size, but QUIC requires a different approach. Section 14 of + [QUIC-TRANSPORT] advises senders to probe larger sizes using DPLPMTUD + [DPLPMTUD] or Path Maximum Transmission Unit Discovery (PMTUD) + [RFC1191] [RFC8201]. This mechanism encourages senders to approach + the maximum packet size, which could then cause fragmentation within + a network segment of which they may not be aware. + + If path performance is limited when forwarding larger packets, an on- + path device should support a maximum packet size for a specific + transport flow and then consistently drop all packets that exceed the + configured size when the inner IPv4 packet has DF set or IPv6 is + used. + + Networks with configurations that would lead to fragmentation of + large packets within a network segment should drop such packets + rather than fragmenting them. Network operators who plan to + implement a more selective policy may start by focusing on QUIC. + + QUIC flows cannot always be easily distinguished from other UDP + traffic, but we assume at least some portion of QUIC traffic can be + identified (see Section 3.1). For networks supporting QUIC, it is + recommended that a path drops any packet larger than the + fragmentation size. When a QUIC endpoint uses DPLPMTUD, it will use + a QUIC probe packet to discover the PMTU. If this probe is lost, it + will not impact the flow of QUIC data. + + IPv4 routers generate an ICMP message when a packet is dropped + because the link MTU was exceeded. [RFC8504] specifies how an IPv6 + node generates an ICMPv6 PTB in this case. PMTUD relies upon an + endpoint receiving such PTB messages [RFC8201], whereas DPLPMTUD does + not reply upon these messages, but can still optionally use these to + improve performance Section 4.6 of [DPLPMTUD]. + + A network cannot know in advance which discovery method is used by a + QUIC endpoint, so it should send a PTB message in addition to + dropping an oversized packet. A generated PTB message should be + compliant with the validation requirements of Section 14.2.1 of + [QUIC-TRANSPORT], otherwise it will be ignored for PMTU discovery. + This provides a signal to the endpoint to prevent the packet size + from growing too large, which can entirely avoid network segment + fragmentation for that flow. + + Endpoints can cache PMTU information in the IP-layer cache. This + short-term consistency between the PMTU for flows can help avoid an + endpoint using a PMTU that is inefficient. The IP cache can also + influence the PMTU value of other IP flows that use the same path + [RFC8201] [DPLPMTUD], including IP packets carrying protocols other + than QUIC. The representation of an IP path is implementation + specific [RFC8201]. + +5. IANA Considerations + + This document has no actions for IANA. + +6. Security Considerations + + QUIC is an encrypted and authenticated transport. That means once + the cryptographic handshake is complete, QUIC endpoints discard most + packets that are not authenticated, greatly limiting the ability of + an attacker to interfere with existing connections. + + However, some information is still observable as supporting + manageability of QUIC traffic inherently involves trade-offs with the + confidentiality of QUIC's control information; this entire document + is therefore security-relevant. + + More security considerations for QUIC are discussed in + [QUIC-TRANSPORT] and [QUIC-TLS], which generally consider active or + passive attackers in the network as well as attacks on specific QUIC + mechanism. + + Version Negotiation packets do not contain any mechanism to prevent + version downgrade attacks. However, future versions of QUIC that use + Version Negotiation packets are required to define a mechanism that + is robust against version downgrade attacks. Therefore, a network + node should not attempt to impact version selection, as version + downgrade may result in connection failure. + +7. References + +7.1. Normative References + + [QUIC-TLS] Thomson, M., Ed. and S. Turner, Ed., "Using TLS to Secure + QUIC", RFC 9001, DOI 10.17487/RFC9001, May 2021, + <https://www.rfc-editor.org/info/rfc9001>. + + [QUIC-TRANSPORT] + Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + +7.2. Informative References + + [DOTS-ARCH] + Mortensen, A., Ed., Reddy.K, T., Ed., Andreasen, F., + Teague, N., and R. Compton, "DDoS Open Threat Signaling + (DOTS) Architecture", RFC 8811, DOI 10.17487/RFC8811, + August 2020, <https://www.rfc-editor.org/info/rfc8811>. + + [DPLPMTUD] Fairhurst, G., Jones, T., Tรผxen, M., Rรผngeler, I., and T. + Vรถlker, "Packetization Layer Path MTU Discovery for + Datagram Transports", RFC 8899, DOI 10.17487/RFC8899, + September 2020, <https://www.rfc-editor.org/info/rfc8899>. + + [IPIM] Allman, M., Beverly, R., and B. Trammell, "Principles for + Measurability in Protocol Design", 9 December 2016, + <https://arxiv.org/abs/1612.02902>. + + [QUIC-APPLICABILITY] + Kรผhlewind, M. and B. Trammell, "Applicability of the QUIC + Transport Protocol", RFC 9308, DOI 10.17487/RFC9308, + September 2022, <https://www.rfc-editor.org/info/rfc9308>. + + [QUIC-GREASE] + Thomson, M., "Greasing the QUIC Bit", RFC 9287, + DOI 10.17487/RFC9287, August 2022, + <https://www.rfc-editor.org/info/rfc9287>. + + [QUIC-HTTP] + Bishop, M., Ed., "HTTP/3", RFC 9114, DOI 10.17487/RFC9114, + June 2022, <https://www.rfc-editor.org/info/rfc9114>. + + [QUIC-INVARIANTS] + Thomson, M., "Version-Independent Properties of QUIC", + RFC 8999, DOI 10.17487/RFC8999, May 2021, + <https://www.rfc-editor.org/info/rfc8999>. + + [QUIC-LB] Duke, M., Banks, N., and C. Huitema, "QUIC-LB: Generating + Routable QUIC Connection IDs", Work in Progress, Internet- + Draft, draft-ietf-quic-load-balancers-14, 11 July 2022, + <https://datatracker.ietf.org/doc/html/draft-ietf-quic- + load-balancers-14>. + + [QUIC-RECOVERY] + Iyengar, J., Ed. and I. Swett, Ed., "QUIC Loss Detection + and Congestion Control", RFC 9002, DOI 10.17487/RFC9002, + May 2021, <https://www.rfc-editor.org/info/rfc9002>. + + [QUIC-RETRY] + Duke, M. and N. Banks, "QUIC Retry Offload", Work in + Progress, Internet-Draft, draft-ietf-quic-retry-offload- + 00, 25 May 2022, <https://datatracker.ietf.org/doc/html/ + draft-ietf-quic-retry-offload-00>. + + [QUIC-TIMEOUT] + Roskind, J., "QUIC", IETF-88 TSV Area Presentation, 7 + November 2013, + <https://www.ietf.org/proceedings/88/slides/slides-88- + tsvarea-10.pdf>. + + [QUIC-VERSION-NEGOTIATION] + Schinazi, D. and E. Rescorla, "Compatible Version + Negotiation for QUIC", Work in Progress, Internet-Draft, + draft-ietf-quic-version-negotiation-10, 27 September 2022, + <https://datatracker.ietf.org/doc/html/draft-ietf-quic- + version-negotiation-10>. + + [RFC1191] Mogul, J. and S. Deering, "Path MTU discovery", RFC 1191, + DOI 10.17487/RFC1191, November 1990, + <https://www.rfc-editor.org/info/rfc1191>. + + [RFC1812] Baker, F., Ed., "Requirements for IP Version 4 Routers", + RFC 1812, DOI 10.17487/RFC1812, June 1995, + <https://www.rfc-editor.org/info/rfc1812>. + + [RFC2475] Blake, S., Black, D., Carlson, M., Davies, E., Wang, Z., + and W. Weiss, "An Architecture for Differentiated + Services", RFC 2475, DOI 10.17487/RFC2475, December 1998, + <https://www.rfc-editor.org/info/rfc2475>. + + [RFC3168] Ramakrishnan, K., Floyd, S., and D. Black, "The Addition + of Explicit Congestion Notification (ECN) to IP", + RFC 3168, DOI 10.17487/RFC3168, September 2001, + <https://www.rfc-editor.org/info/rfc3168>. + + [RFC3449] Balakrishnan, H., Padmanabhan, V., Fairhurst, G., and M. + Sooriyabandara, "TCP Performance Implications of Network + Path Asymmetry", BCP 69, RFC 3449, DOI 10.17487/RFC3449, + December 2002, <https://www.rfc-editor.org/info/rfc3449>. + + [RFC4443] Conta, A., Deering, S., and M. Gupta, Ed., "Internet + Control Message Protocol (ICMPv6) for the Internet + Protocol Version 6 (IPv6) Specification", STD 89, + RFC 4443, DOI 10.17487/RFC4443, March 2006, + <https://www.rfc-editor.org/info/rfc4443>. + + [RFC4459] Savola, P., "MTU and Fragmentation Issues with In-the- + Network Tunneling", RFC 4459, DOI 10.17487/RFC4459, April + 2006, <https://www.rfc-editor.org/info/rfc4459>. + + [RFC4787] Audet, F., Ed. and C. Jennings, "Network Address + Translation (NAT) Behavioral Requirements for Unicast + UDP", BCP 127, RFC 4787, DOI 10.17487/RFC4787, January + 2007, <https://www.rfc-editor.org/info/rfc4787>. + + [RFC4987] Eddy, W., "TCP SYN Flooding Attacks and Common + Mitigations", RFC 4987, DOI 10.17487/RFC4987, August 2007, + <https://www.rfc-editor.org/info/rfc4987>. + + [RFC5382] Guha, S., Ed., Biswas, K., Ford, B., Sivakumar, S., and P. + Srisuresh, "NAT Behavioral Requirements for TCP", BCP 142, + RFC 5382, DOI 10.17487/RFC5382, October 2008, + <https://www.rfc-editor.org/info/rfc5382>. + + [RFC6066] Eastlake 3rd, D., "Transport Layer Security (TLS) + Extensions: Extension Definitions", RFC 6066, + DOI 10.17487/RFC6066, January 2011, + <https://www.rfc-editor.org/info/rfc6066>. + + [RFC7301] Friedl, S., Popov, A., Langley, A., and E. Stephan, + "Transport Layer Security (TLS) Application-Layer Protocol + Negotiation Extension", RFC 7301, DOI 10.17487/RFC7301, + July 2014, <https://www.rfc-editor.org/info/rfc7301>. + + [RFC7605] Touch, J., "Recommendations on Using Assigned Transport + Port Numbers", BCP 165, RFC 7605, DOI 10.17487/RFC7605, + August 2015, <https://www.rfc-editor.org/info/rfc7605>. + + [RFC7801] Dolmatov, V., Ed., "GOST R 34.12-2015: Block Cipher + "Kuznyechik"", RFC 7801, DOI 10.17487/RFC7801, March 2016, + <https://www.rfc-editor.org/info/rfc7801>. + + [RFC7838] Nottingham, M., McManus, P., and J. Reschke, "HTTP + Alternative Services", RFC 7838, DOI 10.17487/RFC7838, + April 2016, <https://www.rfc-editor.org/info/rfc7838>. + + [RFC7983] Petit-Huguenin, M. and G. Salgueiro, "Multiplexing Scheme + Updates for Secure Real-time Transport Protocol (SRTP) + Extension for Datagram Transport Layer Security (DTLS)", + RFC 7983, DOI 10.17487/RFC7983, September 2016, + <https://www.rfc-editor.org/info/rfc7983>. + + [RFC8201] McCann, J., Deering, S., Mogul, J., and R. Hinden, Ed., + "Path MTU Discovery for IP version 6", STD 87, RFC 8201, + DOI 10.17487/RFC8201, July 2017, + <https://www.rfc-editor.org/info/rfc8201>. + + [RFC8504] Chown, T., Loughney, J., and T. Winters, "IPv6 Node + Requirements", BCP 220, RFC 8504, DOI 10.17487/RFC8504, + January 2019, <https://www.rfc-editor.org/info/rfc8504>. + + [RFC9065] Fairhurst, G. and C. Perkins, "Considerations around + Transport Header Confidentiality, Network Operations, and + the Evolution of Internet Transport Protocols", RFC 9065, + DOI 10.17487/RFC9065, July 2021, + <https://www.rfc-editor.org/info/rfc9065>. + + [RFC9250] Huitema, C., Dickinson, S., and A. Mankin, "DNS over + Dedicated QUIC Connections", RFC 9250, + DOI 10.17487/RFC9250, May 2022, + <https://www.rfc-editor.org/info/rfc9250>. + + [TLS-ECH] Rescorla, E., Oku, K., Sullivan, N., and C. A. Wood, "TLS + Encrypted Client Hello", Work in Progress, Internet-Draft, + draft-ietf-tls-esni-14, 13 February 2022, + <https://datatracker.ietf.org/doc/html/draft-ietf-tls- + esni-14>. + + [TMA-QOF] Trammell, B., Gugelmann, D., and N. Brownlee, "Inline Data + Integrity Signals for Passive Measurement", Traffic + Measurement and Analysis, TMA 2014, Lecture Notes in + Computer Science, vol. 8406, pp. 15-25, + DOI 10.1007/978-3-642-54999-1_2, April 2014, + <https://link.springer.com/ + chapter/10.1007/978-3-642-54999-1_2>. + + [WIRE-IMAGE] + Trammell, B. and M. Kuehlewind, "The Wire Image of a + Network Protocol", RFC 8546, DOI 10.17487/RFC8546, April + 2019, <https://www.rfc-editor.org/info/rfc8546>. + +Acknowledgments + + Special thanks to last call reviewers Elwyn Davies, Barry Leiba, Al + Morton, and Peter Saint-Andre. + + This work was partially supported by the European Commission under + Horizon 2020 grant agreement no. 688421 Measurement and Architecture + for a Middleboxed Internet (MAMI), and by the Swiss State Secretariat + for Education, Research, and Innovation under contract no. 15.0268. + This support does not imply endorsement. + +Contributors + + The following people have contributed significant text to and/or + feedback on this document: + + Chris Box + + + Dan Druta + + + David Schinazi + + + Gorry Fairhurst + + + Ian Swett + + + Igor Lubashev + + + Jana Iyengar + + + Jared Mauch + + + Lars Eggert + + + Lucas Purdue + + + Marcus Ihlar + + + Mark Nottingham + + + Martin Duke + + + Martin Thomson + + + Matt Joras + + + Mike Bishop + + + Nick Banks + + + Thomas Fossati + + + Sean Turner + + +Authors' Addresses + + Mirja Kรผhlewind + Ericsson + Email: mirja.kuehlewind@ericsson.com + + + Brian Trammell + Google Switzerland GmbH + Gustav-Gull-Platz 1 + CH-8004 Zurich + Switzerland + Email: ietf@trammell.ch diff --git a/eval/corpora/rfc/RFC 9369 - QUIC Version 2.txt b/eval/corpora/rfc/RFC 9369 - QUIC Version 2.txt new file mode 100644 index 00000000..e44386a7 --- /dev/null +++ b/eval/corpora/rfc/RFC 9369 - QUIC Version 2.txt @@ -0,0 +1,650 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) M. Duke +Request for Comments: 9369 Google LLC +Category: Standards Track May 2023 +ISSN: 2070-1721 + + + QUIC Version 2 + +Abstract + + This document specifies QUIC version 2, which is identical to QUIC + version 1 except for some trivial details. Its purpose is to combat + various ossification vectors and exercise the version negotiation + framework. It also serves as a template for the minimum changes in + any future version of QUIC. + + Note that "version 2" is an informal name for this proposal that + indicates it is the second version of QUIC to be published as a + Standards Track document. The protocol specified here uses a version + number other than 2 in the wire image, in order to minimize + ossification risks. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9369. + +Copyright Notice + + Copyright (c) 2023 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 2. Conventions + 3. Differences with QUIC Version 1 + 3.1. Version Field + 3.2. Long Header Packet Types + 3.3. Cryptography Changes + 3.3.1. Initial Salt + 3.3.2. HMAC-based Key Derivation Function (HKDF) Labels + 3.3.3. Retry Integrity Tag + 4. Version Negotiation Considerations + 4.1. Compatible Negotiation Requirements + 5. TLS Resumption and NEW_TOKEN Tokens + 6. Ossification Considerations + 7. Applicability + 8. Security Considerations + 9. IANA Considerations + 10. References + 10.1. Normative References + 10.2. Informative References + Appendix A. Sample Packet Protection + A.1. Keys + A.2. Client Initial + A.3. Server Initial + A.4. Retry + A.5. ChaCha20-Poly1305 Short Header Packet + Acknowledgments + Author's Address + +1. Introduction + + QUIC version 1 [QUIC] has numerous extension points, including the + version number that occupies the second through fifth bytes of every + long header (see [QUIC-INVARIANTS]). If experimental versions are + rare, and QUIC version 1 constitutes the vast majority of QUIC + traffic, there is the potential for middleboxes to ossify on the + version bytes that are usually 0x00000001. + + In QUIC version 1, Initial packets are encrypted with the version- + specific salt, as described in Section 5.2 of [QUIC-TLS]. Protecting + Initial packets in this way allows observers to inspect their + contents, which includes the TLS Client Hello or Server Hello + messages. Again, there is the potential for middleboxes to ossify on + the version 1 key derivation and packet formats. + + Finally, [QUIC-VN] describes two mechanisms endpoints can use to + negotiate which QUIC version to select. The "incompatible" version + negotiation method can support switching from any QUIC version to any + other version with full generality, at the cost of an additional + round trip at the start of the connection. "Compatible" version + negotiation eliminates the round-trip penalty but levies some + restrictions on how much the two versions can differ semantically. + + QUIC version 2 is meant to mitigate ossification concerns and + exercise the version negotiation mechanisms. The changes provide an + example of the minimum set of changes necessary to specify a new QUIC + version. However, note that the choice of the version number on the + wire is randomly chosen instead of "2", and the two bits that + identify each Long Header packet type are different from version 1; + both of these properties are meant to combat ossification and are not + strictly required of a new QUIC version. + + Any endpoint that supports two versions needs to implement version + negotiation to protect against downgrade attacks. + +2. Conventions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + +3. Differences with QUIC Version 1 + + Except for a few differences, QUIC version 2 endpoints MUST implement + the QUIC version 1 specification as described in [QUIC], [QUIC-TLS], + and [QUIC-RECOVERY]. The remainder of this section lists the + differences. + +3.1. Version Field + + The Version field of long headers is 0x6b3343cf. This was generated + by taking the first four bytes of the sha256sum of "QUICv2 version + number". + +3.2. Long Header Packet Types + + All version 2 Long Header packet types are different. The Type field + values are: + + * Initial: 0b01 + + * 0-RTT: 0b10 + + * Handshake: 0b11 + + * Retry: 0b00 + +3.3. Cryptography Changes + +3.3.1. Initial Salt + + The salt used to derive Initial keys in Section 5.2 of [QUIC-TLS] + changes to: + + initial_salt = 0x0dede3def700a6db819381be6e269dcbf9bd2ed9 + + This is the first 20 bytes of the sha256sum of "QUICv2 salt". + +3.3.2. HMAC-based Key Derivation Function (HKDF) Labels + + The labels used in [QUIC-TLS] to derive packet protection keys + (Section 5.1), header protection keys (Section 5.4), Retry Integrity + Tag keys (Section 5.8), and key updates (Section 6.1) change from + "quic key" to "quicv2 key", from "quic iv" to "quicv2 iv", from + "quic hp" to "quicv2 hp", and from "quic ku" to "quicv2 ku" to meet + the guidance for new versions in Section 9.6 of that document. + +3.3.3. Retry Integrity Tag + + The key and nonce used for the Retry Integrity Tag (Section 5.8 of + [QUIC-TLS]) change to: + + secret = + 0xc4dd2484d681aefa4ff4d69c2c20299984a765a5d3c31982f38fc74162155e9f + key = 0x8fb4b01b56ac48e260fbcbcead7ccc92 + nonce = 0xd86969bc2d7c6d9990efb04a + + The secret is the sha256sum of "QUICv2 retry secret". The key and + nonce are derived from this secret with the labels "quicv2 key" and + "quicv2 iv", respectively. + +4. Version Negotiation Considerations + + QUIC version 2 is not intended to deprecate version 1. Endpoints + that support version 2 might continue support for version 1 to + maximize compatibility with other endpoints. In particular, HTTP + clients often use Alt-Svc [RFC7838] to discover QUIC support. As + this mechanism does not currently distinguish between QUIC versions, + HTTP servers SHOULD support multiple versions to reduce the + probability of incompatibility and the cost associated with QUIC + version negotiation or TCP fallback. For example, an origin + advertising support for "h3" in Alt-Svc should support QUIC version + 1, as it was the original QUIC version used by HTTP/3; therefore, + some clients will only support that version. + + Any QUIC endpoint that supports QUIC version 2 MUST send, process, + and validate the version_information transport parameter specified in + [QUIC-VN] to prevent version downgrade attacks. + + Note that version 2 meets the definition in [QUIC-VN] of a compatible + version with version 1, and version 1 is compatible with version 2. + Therefore, servers can use compatible negotiation to switch a + connection between the two versions. Endpoints that support both + versions SHOULD support compatible version negotiation to avoid a + round trip. + +4.1. Compatible Negotiation Requirements + + Compatible version negotiation between versions 1 and 2 follows the + same requirements in either direction. This section uses the terms + "original version" and "negotiated version" from [QUIC-VN]. + + If the server sends a Retry packet, it MUST use the original version. + The client ignores Retry packets using other versions. The client + MUST NOT use a different version in the subsequent Initial packet + that contains the Retry token. The server MAY encode the QUIC + version in its Retry token to validate that the client did not switch + versions, and drop the packet if it switched, to enforce client + compliance. + + QUIC version 2 uses the same transport parameters to authenticate the + Retry as QUIC version 1. After switching to a negotiated version + after a Retry, the server MUST include the relevant transport + parameters to validate that the server sent the Retry and the + connection IDs used in the exchange, as described in Section 7.3 of + [QUIC]. + + The server cannot send CRYPTO frames until it has processed the + client's transport parameters. The server MUST send all CRYPTO + frames using the negotiated version. + + The client learns the negotiated version by observing the first long + header Version field that differs from the original version. If the + client receives a CRYPTO frame from the server in the original + version, it indicates that the negotiated version is equal to the + original version. + + Before the server is able to process transport parameters from the + client, it might need to respond to Initial packets from the client. + For these packets, the server uses the original version. + + Once the client has learned the negotiated version, it SHOULD send + subsequent Initial packets using that version. The server MUST NOT + discard its original version Initial receive keys until it + successfully processes a Handshake packet with the negotiated + version. + + Both endpoints MUST send Handshake and 1-RTT packets using the + negotiated version. An endpoint MUST drop packets using any other + version. Endpoints have no need to generate the keying material that + would allow them to decrypt or authenticate such packets. + + The client MUST NOT send 0-RTT packets using the negotiated version, + even after processing a packet of that version from the server. + Servers can accept 0-RTT and then process 0-RTT packets from the + original version. + +5. TLS Resumption and NEW_TOKEN Tokens + + TLS session tickets and NEW_TOKEN tokens are specific to the QUIC + version of the connection that provided them. Clients MUST NOT use a + session ticket or token from a QUIC version 1 connection to initiate + a QUIC version 2 connection, and vice versa. When a connection + includes compatible version negotiation, any issued server tokens are + considered to originate from the negotiated version, not the original + one. + + Servers MUST validate the originating version of any session ticket + or token and not accept one issued from a different version. A + rejected ticket results in falling back to a full TLS handshake, + without 0-RTT. A rejected token results in the client address + remaining unverified, which limits the amount of data the server can + send. + + After compatible version negotiation, any resulting session ticket + maps to the negotiated version rather than the original one. + +6. Ossification Considerations + + QUIC version 2 provides protection against some forms of + ossification. Devices that assume that all long headers will encode + version 1, or that the version 1 Initial key derivation formula will + remain version-invariant, will not correctly process version 2 + packets. + + However, many middleboxes, such as firewalls, focus on the first + packet in a connection, which will often remain in the version 1 + format due to the considerations above. + + Clients interested in combating middlebox ossification can initiate a + connection using version 2 if they are reasonably certain the server + supports it and if they are willing to suffer a round-trip penalty if + they are incorrect. In particular, a server that issues a session + ticket for version 2 indicates an intent to maintain version 2 + support while the ticket remains valid, even if support cannot be + guaranteed. + +7. Applicability + + QUIC version 2 provides no change from QUIC version 1 for the + capabilities available to applications. Therefore, all Application- + Layer Protocol Negotiation (ALPN) [RFC7301] codepoints specified to + operate over QUIC version 1 can also operate over this version of + QUIC. In particular, both the "h3" [HTTP/3] and "doq" [RFC9250] + ALPNs can operate over QUIC version 2. + + Unless otherwise stated, all QUIC extensions defined to work with + version 1 also work with version 2. + +8. Security Considerations + + QUIC version 2 introduces no changes to the security or privacy + properties of QUIC version 1. + + The mandatory version negotiation mechanism guards against downgrade + attacks, but downgrades have no security implications, as the version + properties are identical. + + Support for QUIC version 2 can help an observer to fingerprint both + client and server devices. + +9. IANA Considerations + + IANA has added the following entries to the "QUIC Versions" registry + maintained at <https://www.iana.org/assignments/quic>. + + Value: 0x6b3343cf + Status: permanent + Specification: RFC 9369 + Change Controller: IETF + Contact: QUIC WG + + Value: 0x709a50c4 + Status: provisional + Specification: RFC 9369 + Change Controller: IETF + Contact: QUIC WG + Notes: QUIC v2 draft codepoint + +10. References + +10.1. Normative References + + [QUIC] Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [QUIC-RECOVERY] + Iyengar, J., Ed. and I. Swett, Ed., "QUIC Loss Detection + and Congestion Control", RFC 9002, DOI 10.17487/RFC9002, + May 2021, <https://www.rfc-editor.org/info/rfc9002>. + + [QUIC-TLS] Thomson, M., Ed. and S. Turner, Ed., "Using TLS to Secure + QUIC", RFC 9001, DOI 10.17487/RFC9001, May 2021, + <https://www.rfc-editor.org/info/rfc9001>. + + [QUIC-VN] Schinazi, D. and E. Rescorla, "Compatible Version + Negotiation for QUIC", RFC 9368, DOI 10.17487/RFC9368, May + 2023, <https://www.rfc-editor.org/info/rfc9368>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + +10.2. Informative References + + [HTTP/3] Bishop, M., Ed., "HTTP/3", RFC 9114, DOI 10.17487/RFC9114, + June 2022, <https://www.rfc-editor.org/info/rfc9114>. + + [QUIC-INVARIANTS] + Thomson, M., "Version-Independent Properties of QUIC", + RFC 8999, DOI 10.17487/RFC8999, May 2021, + <https://www.rfc-editor.org/info/rfc8999>. + + [RFC7301] Friedl, S., Popov, A., Langley, A., and E. Stephan, + "Transport Layer Security (TLS) Application-Layer Protocol + Negotiation Extension", RFC 7301, DOI 10.17487/RFC7301, + July 2014, <https://www.rfc-editor.org/info/rfc7301>. + + [RFC7838] Nottingham, M., McManus, P., and J. Reschke, "HTTP + Alternative Services", RFC 7838, DOI 10.17487/RFC7838, + April 2016, <https://www.rfc-editor.org/info/rfc7838>. + + [RFC9250] Huitema, C., Dickinson, S., and A. Mankin, "DNS over + Dedicated QUIC Connections", RFC 9250, + DOI 10.17487/RFC9250, May 2022, + <https://www.rfc-editor.org/info/rfc9250>. + +Appendix A. Sample Packet Protection + + This section shows examples of packet protection so that + implementations can be verified incrementally. Samples of Initial + packets from both the client and server plus a Retry packet are + defined. These packets use an 8-byte client-chosen Destination + Connection ID of 0x8394c8f03e515708. Some intermediate values are + included. All values are shown in hexadecimal. + +A.1. Keys + + The labels generated during the execution of the HKDF-Expand-Label + function (that is, HkdfLabel.label) and part of the value given to + the HKDF-Expand function in order to produce its output are: + + client in: 00200f746c73313320636c69656e7420696e00 + + server in: 00200f746c7331332073657276657220696e00 + + quicv2 key: 001010746c73313320717569637632206b657900 + + quicv2 iv: 000c0f746c7331332071756963763220697600 + + quicv2 hp: 00100f746c7331332071756963763220687000 + + The initial secret is common: + + initial_secret = HKDF-Extract(initial_salt, cid) + = 2062e8b3cd8d52092614b8071d0aa1fb + 7c2e3ac193f78b280e72d8f5751f6aba + + The secrets for protecting client packets are: + + client_initial_secret + = HKDF-Expand-Label(initial_secret, "client in", "", 32) + = 14ec9d6eb9fd7af83bf5a668bc17a7e2 + 83766aade7ecd0891f70f9ff7f4bf47b + + key = HKDF-Expand-Label(client_initial_secret, "quicv2 key", "", 16) + = 8b1a0bc121284290a29e0971b5cd045d + + iv = HKDF-Expand-Label(client_initial_secret, "quicv2 iv", "", 12) + = 91f73e2351d8fa91660e909f + + hp = HKDF-Expand-Label(client_initial_secret, "quicv2 hp", "", 16) + = 45b95e15235d6f45a6b19cbcb0294ba9 + + The secrets for protecting server packets are: + + server_initial_secret + = HKDF-Expand-Label(initial_secret, "server in", "", 32) + = 0263db1782731bf4588e7e4d93b74639 + 07cb8cd8200b5da55a8bd488eafc37c1 + + key = HKDF-Expand-Label(server_initial_secret, "quicv2 key", "", 16) + = 82db637861d55e1d011f19ea71d5d2a7 + + iv = HKDF-Expand-Label(server_initial_secret, "quicv2 iv", "", 12) + = dd13c276499c0249d3310652 + + hp = HKDF-Expand-Label(server_initial_secret, "quicv2 hp", "", 16) + = edf6d05c83121201b436e16877593c3a + +A.2. Client Initial + + The client sends an Initial packet. The unprotected payload of this + packet contains the following CRYPTO frame, plus enough PADDING + frames to make a 1162-byte payload: + + 060040f1010000ed0303ebf8fa56f129 39b9584a3896472ec40bb863cfd3e868 + 04fe3a47f06a2b69484c000004130113 02010000c000000010000e00000b6578 + 616d706c652e636f6dff01000100000a 00080006001d00170018001000070005 + 04616c706e0005000501000000000033 00260024001d00209370b2c9caa47fba + baf4559fedba753de171fa71f50f1ce1 5d43e994ec74d748002b000302030400 + 0d0010000e0403050306030203080408 050806002d00020101001c0002400100 + 3900320408ffffffffffffffff050480 00ffff07048000ffff08011001048000 + 75300901100f088394c8f03e51570806 048000ffff + + The unprotected header indicates a length of 1182 bytes: the 4-byte + packet number, 1162 bytes of frames, and the 16-byte authentication + tag. The header includes the connection ID and a packet number of 2: + + d36b3343cf088394c8f03e5157080000449e00000002 + + Protecting the payload produces an output that is sampled for header + protection. Because the header uses a 4-byte packet number encoding, + the first 16 bytes of the protected payload is sampled and then + applied to the header as follows: + + sample = ffe67b6abcdb4298b485dd04de806071 + + mask = AES-ECB(hp, sample)[0..4] + = 94a0c95e80 + + header[0] ^= mask[0] & 0x0f + = d7 + header[18..21] ^= mask[1..4] + = a0c95e82 + header = d76b3343cf088394c8f03e5157080000449ea0c95e82 + + The resulting protected packet is: + + d76b3343cf088394c8f03e5157080000 449ea0c95e82ffe67b6abcdb4298b485 + dd04de806071bf03dceebfa162e75d6c 96058bdbfb127cdfcbf903388e99ad04 + 9f9a3dd4425ae4d0992cfff18ecf0fdb 5a842d09747052f17ac2053d21f57c5d + 250f2c4f0e0202b70785b7946e992e58 a59ac52dea6774d4f03b55545243cf1a + 12834e3f249a78d395e0d18f4d766004 f1a2674802a747eaa901c3f10cda5500 + cb9122faa9f1df66c392079a1b40f0de 1c6054196a11cbea40afb6ef5253cd68 + 18f6625efce3b6def6ba7e4b37a40f77 32e093daa7d52190935b8da58976ff33 + 12ae50b187c1433c0f028edcc4c2838b 6a9bfc226ca4b4530e7a4ccee1bfa2a3 + d396ae5a3fb512384b2fdd851f784a65 e03f2c4fbe11a53c7777c023462239dd + 6f7521a3f6c7d5dd3ec9b3f233773d4b 46d23cc375eb198c63301c21801f6520 + bcfb7966fc49b393f0061d974a2706df 8c4a9449f11d7f3d2dcbb90c6b877045 + 636e7c0c0fe4eb0f697545460c806910 d2c355f1d253bc9d2452aaa549e27a1f + ac7cf4ed77f322e8fa894b6a83810a34 b361901751a6f5eb65a0326e07de7c12 + 16ccce2d0193f958bb3850a833f7ae43 2b65bc5a53975c155aa4bcb4f7b2c4e5 + 4df16efaf6ddea94e2c50b4cd1dfe060 17e0e9d02900cffe1935e0491d77ffb4 + fdf85290fdd893d577b1131a610ef6a5 c32b2ee0293617a37cbb08b847741c3b + 8017c25ca9052ca1079d8b78aebd4787 6d330a30f6a8c6d61dd1ab5589329de7 + 14d19d61370f8149748c72f132f0fc99 f34d766c6938597040d8f9e2bb522ff9 + 9c63a344d6a2ae8aa8e51b7b90a4a806 105fcbca31506c446151adfeceb51b91 + abfe43960977c87471cf9ad4074d30e1 0d6a7f03c63bd5d4317f68ff325ba3bd + 80bf4dc8b52a0ba031758022eb025cdd 770b44d6d6cf0670f4e990b22347a7db + 848265e3e5eb72dfe8299ad7481a4083 22cac55786e52f633b2fb6b614eaed18 + d703dd84045a274ae8bfa73379661388 d6991fe39b0d93debb41700b41f90a15 + c4d526250235ddcd6776fc77bc97e7a4 17ebcb31600d01e57f32162a8560cacc + 7e27a096d37a1a86952ec71bd89a3e9a 30a2a26162984d7740f81193e8238e61 + f6b5b984d4d3dfa033c1bb7e4f0037fe bf406d91c0dccf32acf423cfa1e70710 + 10d3f270121b493ce85054ef58bada42 310138fe081adb04e2bd901f2f13458b + 3d6758158197107c14ebb193230cd115 7380aa79cae1374a7c1e5bbcb80ee23e + 06ebfde206bfb0fcbc0edc4ebec30966 1bdd908d532eb0c6adc38b7ca7331dce + 8dfce39ab71e7c32d318d136b6100671 a1ae6a6600e3899f31f0eed19e3417d1 + 34b90c9058f8632c798d4490da498730 7cba922d61c39805d072b589bd52fdf1 + e86215c2d54e6670e07383a27bbffb5a ddf47d66aa85a0c6f9f32e59d85a44dd + 5d3b22dc2be80919b490437ae4f36a0a e55edf1d0b5cb4e9a3ecabee93dfc6e3 + 8d209d0fa6536d27a5d6fbb17641cde2 7525d61093f1b28072d111b2b4ae5f89 + d5974ee12e5cf7d5da4d6a31123041f3 3e61407e76cffcdcfd7e19ba58cf4b53 + 6f4c4938ae79324dc402894b44faf8af bab35282ab659d13c93f70412e85cb19 + 9a37ddec600545473cfb5a05e08d0b20 9973b2172b4d21fb69745a262ccde96b + a18b2faa745b6fe189cf772a9f84cbfc + +A.3. Server Initial + + The server sends the following payload in response, including an ACK + frame, a CRYPTO frame, and no PADDING frames: + + 02000000000600405a020000560303ee fce7f7b37ba1d1632e96677825ddf739 + 88cfc79825df566dc5430b9a045a1200 130100002e00330024001d00209d3c94 + 0d89690b84d08a60993c144eca684d10 81287c834d5311bcf32bb9da1a002b00 + 020304 + + The header from the server includes a new connection ID and a 2-byte + packet number encoding for a packet number of 1: + + d16b3343cf0008f067a5502a4262b50040750001 + + As a result, after protection, the header protection sample is taken, + starting from the third protected byte: + + sample = 6f05d8a4398c47089698baeea26b91eb + mask = 4dd92e91ea + header = dc6b3343cf0008f067a5502a4262b5004075d92f + + The final protected packet is then: + + dc6b3343cf0008f067a5502a4262b500 4075d92faaf16f05d8a4398c47089698 + baeea26b91eb761d9b89237bbf872630 17915358230035f7fd3945d88965cf17 + f9af6e16886c61bfc703106fbaf3cb4c fa52382dd16a393e42757507698075b2 + c984c707f0a0812d8cd5a6881eaf21ce da98f4bd23f6fe1a3e2c43edd9ce7ca8 + 4bed8521e2e140 + +A.4. Retry + + This shows a Retry packet that might be sent in response to the + Initial packet in Appendix A.2. The integrity check includes the + client-chosen connection ID value of 0x8394c8f03e515708, but that + value is not included in the final Retry packet: + + cf6b3343cf0008f067a5502a4262b574 6f6b656ec8646ce8bfe33952d9555436 + 65dcc7b6 + +A.5. ChaCha20-Poly1305 Short Header Packet + + This example shows some of the steps required to protect a packet + with a short header. It uses AEAD_CHACHA20_POLY1305. + + In this example, TLS produces an application write secret from which + a server uses HKDF-Expand-Label to produce four values: a key, an + Initialization Vector (IV), a header protection key, and the secret + that will be used after keys are updated (this last value is not used + further in this example). + + secret + = 9ac312a7f877468ebe69422748ad00a1 + 5443f18203a07d6060f688f30f21632b + + key = HKDF-Expand-Label(secret, "quicv2 key", "", 32) + = 3bfcddd72bcf02541d7fa0dd1f5f9eee + a817e09a6963a0e6c7df0f9a1bab90f2 + + iv = HKDF-Expand-Label(secret, "quicv2 iv", "", 12) + = a6b5bc6ab7dafce30ffff5dd + + hp = HKDF-Expand-Label(secret, "quicv2 hp", "", 32) + = d659760d2ba434a226fd37b35c69e2da + 8211d10c4f12538787d65645d5d1b8e2 + + ku = HKDF-Expand-Label(secret, "quicv2 ku", "", 32) + = c69374c49e3d2a9466fa689e49d476db + 5d0dfbc87d32ceeaa6343fd0ae4c7d88 + + The following shows the steps involved in protecting a minimal packet + with an empty Destination Connection ID. This packet contains a + single PING frame (that is, a payload of just 0x01) and has a packet + number of 654360564. In this example, using a packet number of + length 3 (that is, 49140 is encoded) avoids having to pad the payload + of the packet; PADDING frames would be needed if the packet number is + encoded on fewer bytes. + + pn = 654360564 (decimal) + nonce = a6b5bc6ab7dafce328ff4a29 + unprotected header = 4200bff4 + payload plaintext = 01 + payload ciphertext = 0ae7b6b932bc27d786f4bc2bb20f2162ba + + The resulting ciphertext is the minimum size possible. One byte is + skipped to produce the sample for header protection. + + sample = e7b6b932bc27d786f4bc2bb20f2162ba + mask = 97580e32bf + header = 5558b1c6 + + The protected packet is the smallest possible packet size of 21 + bytes. + + packet = 5558b1c60ae7b6b932bc27d786f4bc2bb20f2162ba + +Acknowledgments + + The author would like to thank Christian Huitema, Lucas Pardue, Kyle + Rose, Anthony Rossi, Zahed Sarker, David Schinazi, Tatsuhiro + Tsujikawa, and Martin Thomson for their helpful suggestions. + +Author's Address + + Martin Duke + Google LLC + Email: martin.h.duke@gmail.com diff --git a/eval/corpora/rfc/RFC 9412 - The ORIGIN Extension in HTTP3.txt b/eval/corpora/rfc/RFC 9412 - The ORIGIN Extension in HTTP3.txt new file mode 100644 index 00000000..bd4d4715 --- /dev/null +++ b/eval/corpora/rfc/RFC 9412 - The ORIGIN Extension in HTTP3.txt @@ -0,0 +1,193 @@ +๏ปฟ + + + +Internet Engineering Task Force (IETF) M. Bishop +Request for Comments: 9412 Akamai +Category: Standards Track June 2023 +ISSN: 2070-1721 + + + The ORIGIN Extension in HTTP/3 + +Abstract + + The ORIGIN frame for HTTP/2 is equally applicable to HTTP/3, but it + needs to be separately registered. This document describes the + ORIGIN frame for HTTP/3. + +Status of This Memo + + This is an Internet Standards Track document. + + This document is a product of the Internet Engineering Task Force + (IETF). It represents the consensus of the IETF community. It has + received public review and has been approved for publication by the + Internet Engineering Steering Group (IESG). Further information on + Internet Standards is available in Section 2 of RFC 7841. + + Information about the current status of this document, any errata, + and how to provide feedback on it may be obtained at + https://www.rfc-editor.org/info/rfc9412. + +Copyright Notice + + Copyright (c) 2023 IETF Trust and the persons identified as the + document authors. All rights reserved. + + This document is subject to BCP 78 and the IETF Trust's Legal + Provisions Relating to IETF Documents + (https://trustee.ietf.org/license-info) in effect on the date of + publication of this document. Please review these documents + carefully, as they describe your rights and restrictions with respect + to this document. Code Components extracted from this document must + include Revised BSD License text as described in Section 4.e of the + Trust Legal Provisions and are provided without warranty as described + in the Revised BSD License. + +Table of Contents + + 1. Introduction + 1.1. Notational Conventions + 2. The ORIGIN HTTP/3 Frame + 2.1. Frame Layout + 3. Security Considerations + 4. IANA Considerations + 5. References + 5.1. Normative References + 5.2. Informative References + Author's Address + +1. Introduction + + Existing RFCs define extensions to HTTP/2 [HTTP/2] that remain useful + in HTTP/3. Appendix A.2 of [HTTP/3] describes the required updates + for HTTP/2 frames to be used with HTTP/3. + + [ORIGIN] defines the HTTP/2 ORIGIN frame, which indicates what + origins are available on a given connection. It defines a single + HTTP/2 frame type. + +1.1. Notational Conventions + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + BCP 14 [RFC2119] [RFC8174] when, and only when, they appear in all + capitals, as shown here. + + The frame diagram in this document uses the format defined in + Section 1.3 of [QUIC-TRANSPORT] to illustrate the order and size of + fields. + +2. The ORIGIN HTTP/3 Frame + + The ORIGIN HTTP/3 frame allows a server to indicate what origin or + origins [RFC6454] the server would like the client to consider as one + or more members of the Origin Set (Section 2.3 of [ORIGIN]) for the + connection within which it occurs. + + The semantics of the frame payload are identical to those of the + HTTP/2 frame defined in [ORIGIN]. Where HTTP/2 reserves stream 0 for + frames related to the state of the connection, HTTP/3 defines a pair + of unidirectional streams called "control streams" for this purpose. + + Where [ORIGIN] indicates that the ORIGIN frame is sent on stream 0, + this should be interpreted to mean the HTTP/3 control stream: that + is, the ORIGIN frame is sent from servers to clients on the server's + control stream. + + HTTP/3 does not define a Flags field in the generic frame layout. As + no flags have been defined for the ORIGIN frame, this specification + does not define a mechanism for communicating such flags in HTTP/3. + +2.1. Frame Layout + + The ORIGIN frame has a layout that is nearly identical to the layout + used in HTTP/2; the information is restated here for clarity. The + ORIGIN frame type is 0x0c (decimal 12), as in HTTP/2. The payload + contains zero or more instances of the Origin-Entry field. + + HTTP/3 Origin-Entry { + Origin-Len (16), + ASCII-Origin (..), + } + + HTTP/3 ORIGIN Frame { + Type (i) = 0x0c, + Length (i), + Origin-Entry (..) ..., + } + + Figure 1: ORIGIN Frame Layout + + An Origin-Entry is a length-delimited string. Specifically, it + contains two fields: + + Origin-Len: An unsigned, 16-bit integer indicating the length, in + octets, of the ASCII-Origin field. + + ASCII-Origin: An OPTIONAL sequence of characters containing the + ASCII serialization of an origin ([RFC6454], Section 6.2) that the + sender asserts this connection is or could be authoritative for. + +3. Security Considerations + + This document introduces no new security considerations beyond those + discussed in [ORIGIN] and [HTTP/3]. + +4. IANA Considerations + + This document registers a frame type in the "HTTP/3 Frame Types" + registry defined by [HTTP/3], located at + <https://www.iana.org/assignments/http3-parameters/>. + + Value: 0x0c + Frame Type: ORIGIN + Status: permanent + Reference: Section 2 + Date: 2023-03-14 + Change Controller: IETF + Contact: HTTP WG <ietf-http-wg@w3.org> + +5. References + +5.1. Normative References + + [HTTP/2] Thomson, M., Ed. and C. Benfield, Ed., "HTTP/2", RFC 9113, + DOI 10.17487/RFC9113, June 2022, + <https://www.rfc-editor.org/info/rfc9113>. + + [HTTP/3] Bishop, M., Ed., "HTTP/3", RFC 9114, DOI 10.17487/RFC9114, + June 2022, <https://www.rfc-editor.org/info/rfc9114>. + + [ORIGIN] Nottingham, M. and E. Nygren, "The ORIGIN HTTP/2 Frame", + RFC 8336, DOI 10.17487/RFC8336, March 2018, + <https://www.rfc-editor.org/info/rfc8336>. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, + DOI 10.17487/RFC2119, March 1997, + <https://www.rfc-editor.org/info/rfc2119>. + + [RFC8174] Leiba, B., "Ambiguity of Uppercase vs Lowercase in RFC + 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, + May 2017, <https://www.rfc-editor.org/info/rfc8174>. + +5.2. Informative References + + [QUIC-TRANSPORT] + Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A UDP-Based + Multiplexed and Secure Transport", RFC 9000, + DOI 10.17487/RFC9000, May 2021, + <https://www.rfc-editor.org/info/rfc9000>. + + [RFC6454] Barth, A., "The Web Origin Concept", RFC 6454, + DOI 10.17487/RFC6454, December 2011, + <https://www.rfc-editor.org/info/rfc6454>. + +Author's Address + + Mike Bishop + Akamai + Email: mbishop@evequefou.be diff --git a/eval/corpora/rfc/download.py b/eval/corpora/rfc/download.py new file mode 100644 index 00000000..c1eec398 --- /dev/null +++ b/eval/corpora/rfc/download.py @@ -0,0 +1,158 @@ +"""Re-download the `rfc` corpus: 23 interlinked IETF RFCs (the QUIC / HTTP-3 family). + +This corpus is **not** synthetic and **not** authored by this project. It exists to +answer one question the other corpora cannot: does the pipeline โ€” and in +particular the index-time cross-reference extractor in +``rag_system/indexing/crossref.py`` โ€” behave on documents whose naming and +referencing conventions we did not invent? + +Every file is fetched verbatim from the RFC Editor's canonical plain-text +endpoint ``https://www.rfc-editor.org/rfc/rfcNNNN.txt``. Nothing is edited after +download; only the *filename* is ours, and that choice is deliberate โ€” see +MANIFEST.md ยง "Naming". + + .venv/bin/python eval/corpora/rfc/download.py # fetch anything missing + .venv/bin/python eval/corpora/rfc/download.py --force # re-fetch everything + .venv/bin/python eval/corpora/rfc/download.py --check # verify sizes only + +Selection rule (enforced by ``--check``): every document in the set must +reference, or be referenced by, at least two others in the set. The check is +mechanical โ€” it counts ``RFC NNNN`` / ``[RFCNNNN]`` mentions across the corpus. +""" + +from __future__ import annotations + +import argparse +import os +import re +import sys +import urllib.request +from collections import defaultdict + +HERE = os.path.dirname(os.path.abspath(__file__)) +BASE_URL = "https://www.rfc-editor.org/rfc/rfc{num}.txt" + +# (rfc number, filename title). The filename is +# "RFC <num> - <Title>.txt" +# which is the convention a human filing these on disk would plausibly use, and +# it is what the cross-reference resolver is being tested against. +DOCUMENTS = [ + # --- normative boilerplate that literally every document below cites --- + (2119, "Key Words for Use in RFCs to Indicate Requirement Levels"), + (8174, "Ambiguity of Uppercase vs Lowercase in RFC 2119 Key Words"), + (8126, "Guidelines for Writing an IANA Considerations Section in RFCs"), + # --- TLS-side dependencies of QUIC and HTTP --- + (6066, "TLS Extensions Extension Definitions"), + (7301, "TLS Application-Layer Protocol Negotiation Extension"), + # --- the QUIC core --- + (8999, "Version-Independent Properties of QUIC"), + (9000, "QUIC A UDP-Based Multiplexed and Secure Transport"), + (9001, "Using TLS to Secure QUIC"), + (9002, "QUIC Loss Detection and Congestion Control"), + (9221, "An Unreliable Datagram Extension to QUIC"), + (9369, "QUIC Version 2"), + (9308, "Applicability of the QUIC Transport Protocol"), + (9312, "Manageability of the QUIC Transport Protocol"), + # --- the HTTP-over-QUIC layer --- + (9114, "HTTP3"), + (9204, "QPACK Field Compression for HTTP3"), + (9218, "Extensible Prioritization Scheme for HTTP"), + (9220, "Bootstrapping WebSockets with HTTP3"), + (9297, "HTTP Datagrams and the Capsule Protocol"), + (9298, "Proxying UDP in HTTP"), + (9412, "The ORIGIN Extension in HTTP3"), + # --- the HTTP/2 counterparts the HTTP/3 documents are defined against --- + (8336, "The ORIGIN HTTP2 Frame"), + (8441, "Bootstrapping WebSockets with HTTP2"), + # --- a QUIC application protocol other than HTTP --- + (9250, "DNS over Dedicated QUIC Connections"), +] + +_MENTION_RE = re.compile(r"\bRFC\s?(\d{3,5})\b") + + +def filename(num: int, title: str) -> str: + return f"RFC {num} - {title}.txt" + + +def path_for(num: int, title: str) -> str: + return os.path.join(HERE, filename(num, title)) + + +def fetch(num: int, title: str, force: bool) -> tuple: + target = path_for(num, title) + if os.path.exists(target) and not force: + return ("cached", os.path.getsize(target)) + url = BASE_URL.format(num=num) + with urllib.request.urlopen(url, timeout=60) as response: + payload = response.read() + with open(target, "wb") as fh: + fh.write(payload) + return ("downloaded", len(payload)) + + +def link_graph() -> dict: + """{rfc number: set(other corpus rfc numbers it mentions in its text)}.""" + numbers = {num for num, _ in DOCUMENTS} + graph = {} + for num, title in DOCUMENTS: + target = path_for(num, title) + if not os.path.exists(target): + continue + with open(target, "r", encoding="utf-8", errors="replace") as fh: + text = fh.read() + mentioned = {int(m) for m in _MENTION_RE.findall(text)} + graph[num] = (mentioned & numbers) - {num} + return graph + + +def check() -> int: + graph = link_graph() + missing = [n for n, _ in DOCUMENTS if n not in graph] + if missing: + print(f"MISSING files for: {missing}") + return 1 + + inbound = defaultdict(set) + for source, targets in graph.items(): + for target in targets: + inbound[target].add(source) + + total = 0 + failures = [] + print(f"{'RFC':>6} {'bytes':>8} {'->':>3} {'<-':>3} degree") + for num, title in DOCUMENTS: + size = os.path.getsize(path_for(num, title)) + total += size + out_degree = len(graph[num]) + in_degree = len(inbound[num]) + degree = len(graph[num] | inbound[num]) + print(f"{num:>6} {size:>8} {out_degree:>3} {in_degree:>3} {degree}") + if degree < 2: + failures.append(num) + print(f"\n{len(DOCUMENTS)} documents, {total} bytes ({total / 1024 / 1024:.2f} MiB)") + edges = sum(len(v) for v in graph.values()) + print(f"{edges} directed intra-corpus RFC-number references") + if failures: + print(f"\nFAIL: {failures} are connected to fewer than 2 other documents") + return 1 + print("every document is connected to at least 2 others in the set.") + return 0 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--force", action="store_true", help="re-download files that exist") + parser.add_argument("--check", action="store_true", help="skip downloading; verify only") + args = parser.parse_args() + + if not args.check: + for num, title in DOCUMENTS: + status, size = fetch(num, title, args.force) + print(f" {status:<11} {filename(num, title)} ({size} bytes)") + return check() + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/corpora/rfc/rfc.facts.json b/eval/corpora/rfc/rfc.facts.json new file mode 100644 index 00000000..df33660a --- /dev/null +++ b/eval/corpora/rfc/rfc.facts.json @@ -0,0 +1,189 @@ +{ + "corpus": "rfc", + "documents_dir": ".", + "description": "Answer-bearing verbatim anchors into the 23 RFCs of the `rfc` corpus. Strings that span a hard line break in the RFC's 72-column layout are written on one line here; every gate compares whitespace-normalised text, which is also how eval/run_eval.py scores chunk relevance.", + "facts": [ + { + "id": "q9000_ack_delay_exponent_default", + "topic": "transport_parameters", + "source": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", + "expected": "default value of 3 is assumed (indicating a multiplier of 8).", + "summary": "An absent ack_delay_exponent is taken to be 3, i.e. ACK Delay scales by 8." + }, + { + "id": "q9000_active_cid_limit_floor", + "topic": "transport_parameters", + "source": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", + "expected": "active_connection_id_limit parameter MUST be at least 2.", + "summary": "The advertised active_connection_id_limit may not be below 2." + }, + { + "id": "q9000_no_viable_path", + "topic": "error_codes", + "source": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", + "expected": "NO_VIABLE_PATH (0x10): An endpoint has determined that the network", + "summary": "Transport error code 0x10 is NO_VIABLE_PATH." + }, + { + "id": "q9000_anti_amplification", + "topic": "address_validation", + "source": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", + "expected": "to the unvalidated address to three times the amount of data received", + "summary": "Before address validation an endpoint may send at most 3x what it received." + }, + { + "id": "q9000_initial_max_streams_uni", + "topic": "transport_parameters", + "source": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", + "expected": "initial_max_streams_uni (0x09): The initial maximum unidirectional", + "summary": "Transport parameter 0x09 bounds the peer's unidirectional streams." + }, + { + "id": "q9001_retry_integrity_tag", + "topic": "packet_protection", + "source": "RFC 9001 - Using TLS to Secure QUIC.txt", + "expected": "The Retry Integrity Tag is a 128-bit field that is computed as the", + "summary": "The Retry Integrity Tag is 128 bits wide." + }, + { + "id": "q9001_initial_salt_v1", + "topic": "key_derivation", + "source": "RFC 9001 - Using TLS to Secure QUIC.txt", + "expected": "initial_salt = 0x38762cf7f55934b34d179ae6a4c80cadccbb7f0a", + "summary": "The QUIC version 1 Initial-keys HKDF salt." + }, + { + "id": "q9002_initial_rtt", + "topic": "loss_recovery", + "source": "RFC 9002 - QUIC Loss Detection and Congestion Control.txt", + "expected": "SHOULD be set to 333 milliseconds. This results in handshakes", + "summary": "kInitialRtt defaults to 333 ms before any RTT sample exists." + }, + { + "id": "q9002_pto_probe_count", + "topic": "loss_recovery", + "source": "RFC 9002 - QUIC Loss Detection and Congestion Control.txt", + "expected": "MAY send up to two full-sized datagrams containing ack-eliciting", + "summary": "On PTO expiry a sender may send up to two full-sized probe datagrams." + }, + { + "id": "q9002_pto_formula", + "topic": "loss_recovery", + "source": "RFC 9002 - QUIC Loss Detection and Congestion Control.txt", + "expected": "PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay", + "summary": "The Probe Timeout formula." + }, + { + "id": "q9221_datagram_frame_types", + "topic": "datagram_extension", + "source": "RFC 9221 - An Unreliable Datagram Extension to QUIC.txt", + "expected": "form 0b0011000X (or the values 0x30 and 0x31). The least significant", + "summary": "DATAGRAM frame types are 0x30 and 0x31." + }, + { + "id": "q9369_version_number", + "topic": "version_negotiation", + "source": "RFC 9369 - QUIC Version 2.txt", + "expected": "The Version field of long headers is 0x6b3343cf. This was generated", + "summary": "QUIC version 2's long-header version number." + }, + { + "id": "q9369_initial_salt_v2", + "topic": "key_derivation", + "source": "RFC 9369 - QUIC Version 2.txt", + "expected": "initial_salt = 0x0dede3def700a6db819381be6e269dcbf9bd2ed9", + "summary": "The QUIC version 2 Initial-keys HKDF salt." + }, + { + "id": "q8999_cid_length_range", + "topic": "invariants", + "source": "RFC 8999 - Version-Independent Properties of QUIC.txt", + "expected": "the Destination Connection ID Length field and is between 0 and 255", + "summary": "Version-independently, a Destination Connection ID is 0-255 bytes." + }, + { + "id": "q9250_no_port_53", + "topic": "transport", + "source": "RFC 9250 - DNS over Dedicated QUIC Connections.txt", + "expected": "DoQ connections MUST NOT use UDP port 53.", + "summary": "DNS over QUIC is forbidden on UDP port 53." + }, + { + "id": "q9114_uni_stream_credit", + "topic": "stream_setup", + "source": "RFC 9114 - HTTP3.txt", + "expected": "provide at least 1,024 bytes of flow-control credit to each", + "summary": "HTTP/3 endpoints should grant >=1024 bytes of credit per unidirectional stream." + }, + { + "id": "q9114_three_uni_streams", + "topic": "stream_setup", + "source": "RFC 9114 - HTTP3.txt", + "expected": "Therefore, the transport parameters sent by both clients and servers MUST allow the peer to create at least three", + "summary": "HTTP/3 needs transport parameters permitting at least three unidirectional streams." + }, + { + "id": "q9204_qpack_capacity_default", + "topic": "settings", + "source": "RFC 9204 - QPACK Field Compression for HTTP3.txt", + "expected": "SETTINGS_QPACK_MAX_TABLE_CAPACITY (0x01): The default value is zero.", + "summary": "QPACK dynamic table capacity defaults to zero." + }, + { + "id": "q9297_capsule_forbidden_headers", + "topic": "capsule_protocol", + "source": "RFC 9297 - HTTP Datagrams and the Capsule Protocol.txt", + "expected": "The Capsule Protocol MUST NOT be used with messages that contain Content-Length, Content-Type, or Transfer-Encoding header fields.", + "summary": "Those three header fields disqualify a message from the Capsule Protocol." + }, + { + "id": "q9297_datagram_capsule_type", + "topic": "capsule_protocol", + "source": "RFC 9297 - HTTP Datagrams and the Capsule Protocol.txt", + "expected": "This document defines the DATAGRAM (0x00) Capsule Type.", + "summary": "Capsule Type 0x00 is DATAGRAM." + }, + { + "id": "q9218_urgency_default", + "topic": "priority_parameters", + "source": "RFC 9218 - Extensible Prioritization Scheme for HTTP.txt", + "expected": "between 0 and 7 inclusive, in descending order of priority. The default is 3.", + "summary": "Urgency runs 0-7 and defaults to 3." + }, + { + "id": "q9312_spin_bit_position", + "topic": "observability", + "source": "RFC 9312 - Manageability of the QUIC Transport Protocol.txt", + "expected": "latency spin bit: The third-most-significant bit of the first octet", + "summary": "The latency spin bit is the third-most-significant bit of byte 0." + }, + { + "id": "q7301_alpn_alert_value", + "topic": "alpn", + "source": "RFC 7301 - TLS Application-Layer Protocol Negotiation Extension.txt", + "expected": "no_application_protocol(120),", + "summary": "The no_application_protocol TLS alert description is 120." + }, + { + "id": "q8126_specification_required", + "topic": "registration_policies", + "source": "RFC 8126 - Guidelines for Writing an IANA Considerations Section in RFCs.txt", + "expected": "This policy is the same as Expert Review, with the additional requirement of a formal public specification.", + "summary": "Specification Required = Expert Review plus a permanent public specification." + }, + { + "id": "q8441_protocol_pseudo_header", + "topic": "extended_connect", + "source": "RFC 8441 - Bootstrapping WebSockets with HTTP2.txt", + "expected": "A new pseudo-header field :protocol MAY be included on request", + "summary": "RFC 8441 defines the :protocol pseudo-header for Extended CONNECT." + }, + { + "id": "q8336_origin_frame_type", + "topic": "origin_frame", + "source": "RFC 8336 - The ORIGIN HTTP2 Frame.txt", + "expected": "The ORIGIN frame type is 0xc (decimal 12) and contains zero or more", + "summary": "The HTTP/2 ORIGIN frame is type 0xc." + } + ] +} diff --git a/eval/corpora/verify_facts.py b/eval/corpora/verify_facts.py new file mode 100644 index 00000000..14d59b57 --- /dev/null +++ b/eval/corpora/verify_facts.py @@ -0,0 +1,128 @@ +"""Assert every planted-fact 'expected' string really occurs in its source document. + +This is the first of the two gold-set verification gates. It checks the *source +text*; ``eval/run_eval.py --coverage-only`` checks the second gate โ€” that the +string survives conversion and chunking into at least one indexed chunk. + + .venv/bin/python eval/corpora/verify_facts.py +""" + +import glob +import json +import os +import sys + +HERE = os.path.dirname(os.path.abspath(__file__)) +REPO_ROOT = os.path.abspath(os.path.join(HERE, "..", "..")) + + +def normalise(text: str) -> str: + return " ".join(text.split()) + + +def pdf_text(path: str) -> str: + import pymupdf + + doc = pymupdf.open(path) + try: + return " ".join(page.get_text() for page in doc) + finally: + doc.close() + + +def source_text_for(sidecar: dict, sidecar_path: str) -> dict: + """Return {source_label: normalised_text} for one sidecar file. + + Three shapes, keyed off which field the sidecar carries rather than off the + corpus name, so a new corpus needs no edit here. Relative paths resolve + against the sidecar's own directory: + + * ``source_glob`` โ€” markdown files under the repo root (the ``docs`` corpus) + * ``documents_dir`` โ€” every PDF or plain-text file in that directory + (``acq`` PDFs; the ``rfc`` corpus's 23 RFC Editor .txt files) + * ``document`` โ€” one PDF next to its sidecar (``atlas7``, ``hr``) + """ + here = os.path.dirname(sidecar_path) + if "source_glob" in sidecar: + texts = {} + for path in sorted(glob.glob(os.path.join(REPO_ROOT, sidecar["source_glob"]))): + with open(path, "r", encoding="utf-8") as fh: + texts[os.path.basename(path)] = normalise(fh.read()) + return texts + if "documents_dir" in sidecar: + directory = os.path.join(here, sidecar["documents_dir"]) + texts = {os.path.basename(p): normalise(pdf_text(p)) + for p in sorted(glob.glob(os.path.join(directory, "*.pdf")))} + for path in sorted(glob.glob(os.path.join(directory, "*.txt"))): + with open(path, "r", encoding="utf-8", errors="replace") as fh: + texts[os.path.basename(path)] = normalise(fh.read()) + return texts + path = os.path.join(here, sidecar["document"]) + return {sidecar["document"]: normalise(pdf_text(path))} + + +def check_cross_references(sidecar: dict, texts: dict) -> int: + """Every declared cross-reference cue must really occur in its 'from' document. + + The ``acq`` corpus exists so roadmap items 4.2/4.3 have something to hop + across; a cue that is not literally in the source would make the corpus lie + about its own link graph. ``to: null`` marks a deliberately dangling + reference (the referenced document is not in the corpus) โ€” the cue is still + checked, the target is not. + """ + refs = sidecar.get("cross_references") or [] + if not refs: + return 0 + failures = 0 + dangling = 0 + for ref in refs: + cue = normalise(ref["cue"]) + source = ref["from"] + if source not in texts: + failures += 1 + print(f" MISSING xref source document {source!r}") + continue + if cue not in texts[source]: + failures += 1 + print(f" MISSING xref cue in {source}: {cue!r}") + target = ref.get("to") + if target is None: + dangling += 1 + elif target not in texts: + failures += 1 + print(f" MISSING xref target document {target!r} (from {source})") + print(f" {len(refs) - failures}/{len(refs)} cross-reference cues verified " + f"({dangling} deliberately dangling)") + return failures + + +def main() -> int: + failures = 0 + total = 0 + xref_failures = 0 + for sidecar_path in sorted(glob.glob(os.path.join(HERE, "**", "*.facts.json"), + recursive=True)): + with open(sidecar_path, "r", encoding="utf-8") as fh: + sidecar = json.load(fh) + texts = source_text_for(sidecar, sidecar_path) + joined = " || ".join(texts.values()) + print(f"\n{os.path.basename(sidecar_path)} โ€” corpus '{sidecar['corpus']}', " + f"{len(sidecar['facts'])} facts over {len(texts)} source file(s)") + for fact in sidecar["facts"]: + total += 1 + expected = normalise(fact["expected"]) + named_source = fact.get("source") + haystack = texts.get(named_source, joined) if named_source else joined + if expected not in haystack: + failures += 1 + print(f" MISSING {fact['id']}: {expected!r}" + f"{' in ' + named_source if named_source else ''}") + xref_failures += check_cross_references(sidecar, texts) + print(f"\n{total - failures}/{total} planted facts verified present in their source document.") + if xref_failures: + print(f"{xref_failures} cross-reference cue(s) NOT found in their source document.") + return 1 if (failures or xref_failures) else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/decisions/completeness-clause-ab-2026-08-15.md b/eval/decisions/completeness-clause-ab-2026-08-15.md new file mode 100644 index 00000000..47e74a27 --- /dev/null +++ b/eval/decisions/completeness-clause-ab-2026-08-15.md @@ -0,0 +1,75 @@ +# Completeness-clause A/B (arm J) โ€” 2026-08-15 + +**Verdict: REVERTED.** The clause does not fix the failure mode it was written +for, and its only aggregate movement is inside the documented noise band. + +## What was tested + +Cross-bench validation (eval/decisions/cross-bench-validation-2026-08-15.md) +found hr regressed 24โ†’21 under the strict synthesis prompt because the model +answers the literal question and omits gold clauses attached to the value +("substitute day off", "effective 1 February 2026"). The candidate fix was a +new strict-prompt rule 8: + +> When you state a value, rate, entitlement or rule, also include any +> condition, exception, alternative or effective date that the snippets attach +> directly to it. Answering the literal question while omitting an attached +> qualifier counts as an incomplete answer. + +Per the anti-overfit rule established in the cross-bench record, the change was +gated on the full 5-bench suite (120 rows): rfc gold (arm J, +`rfc_shakedown/run_e2e_arm_j.py`) + the 4 authored corpora (ARM=clause, +`authored_bench/run_e2e_authored.py`). Identical indexes; both arms clean, 0 +errors. + +## Primary evidence: the three target rows + +| row | gold qualifier missing at HEAD | arm J behaviour | +|---|---|---| +| hr_h05 | "2 days extended family" | **Answer byte-identical to HEAD.** Clause had no effect (the missing text is a parallel fact, not a qualifier "attached to the value" โ€” the clause wording never reaches it). | +| hr_h08 | "effective 1 February 2026" | Still missing; answer also *dropped* the source-document mention HEAD included. | +| hr_h13 | "substitute day off within the same quarter" | Still missing. Instead the clause induced padding with irrelevant qualifiers (policy id, owner, Gothenburg, "9 public holidays"). | + +Since hr_h05's answer is byte-identical to one the Sonnet panel already judged +wrong, arm J cannot recover the hr regression regardless of judging. The +clause fails its purpose by construction โ€” no Sonnet panel spend needed for a +non-adoption. + +## Secondary evidence: 4b coarse screen (not the deciding measurement) + +| bench | HEAD (4b) | arm J (4b) | +|---|---|---| +| acq | 18/24 | 16/24 | +| atlas7 | 19/24 | 22/24 | +| hr | 19/24 | 20/24 | +| docs | 13/24 | 14/24 | +| authored total | 69/96 | 72/96 | +| rfc (same-judge I vs J) | 14/24 | 14/24 | + ++3/96 with 16 downward flips and 19 upward flips is churn, not signal +(1โ€“2 rows on n=24 is the established noise floor; this is that, four times). +New losses include hr_h04 โ€” the parental-leave row now omits a qualifier, the +exact failure mode the clause was meant to prevent. Mechanical substring: +docs 17โ†’14, rfc 10โ†’9. Mean answer length ~unchanged (576โ†’555 chars), so the +clause did not even make answers systematically more complete. + +## Interpretation + +The hr โˆ’3 is a *selective-attention* failure, not an instruction-shortage +failure: the model already has the qualifier in context and a rule telling it +to be thorough (rule 6). Adding a more specific instruction moved which rows +it pays attention to, not how completely it answers. Prompt-wording churn that +does not achieve its stated purpose is risk without benefit โ€” and adopting it +anyway on the strength of noise-band movement elsewhere (atlas7 +3 under a +judge documented to contradict itself) would be exactly the +tuned-to-the-bench drift this eval program exists to prevent. + +## Status + +- Rule 8 reverted; strict prompt back to the arm-I form (7 rules). +- hr 24โ†’21 stands as a known, documented cost of the strict prompt + (style, not grounding). Future candidates for it should change *what the + synthesizer attends to* (e.g. qualifier-aware snippet formatting), not add + more prompt rules; any such change re-runs this same 5-bench gate. +- Artifacts: `rfc_e2e_answers_j.jsonl`, `authored_e2e_answers_clause.jsonl`, + `judged4b_clause.jsonl`, `judged4b_rfc_{i,j}.jsonl` in the session scratchpad. diff --git a/eval/decisions/component-ablation-2026-08-18.md b/eval/decisions/component-ablation-2026-08-18.md new file mode 100644 index 00000000..eb677189 --- /dev/null +++ b/eval/decisions/component-ablation-2026-08-18.md @@ -0,0 +1,52 @@ +# Component ablation study โ€” 2026-08-18 + +**Question (user-directed, prompted by the HF multi-vector-encoder post): does +each pipeline component actually help?** Method: six arms, each the full +pipeline minus ONE component, on all five benches (120 rows/arm, v2 indexes, +temp-0 deterministic 4b judge; 3-voter Sonnet panels on every flipped cell of +the actionable arms โ€” 72 cells, 4 split votes). Baseline = arm M/"fixed" +(92/120 4b-det). + +## Results + +| arm | component removed | 4b-det | panel-corrected net | verdict | +|---|---|---|---|---| +| norerank | Qwen3-Reranker-4B selection | 85 (โˆ’7) | not panelled (beyond noise) | **KEEP** โ€” largest single contributor; losses concentrate on the chunk-rich corpora (acq โˆ’4, docs โˆ’2) and it re-loses hr_h05 | +| nodecomp | query decomposition | 95 (+3) | **+3 real** (4 gains, 1 loss, 0 noise) | **HURTING on single-turn** โ€” splitting adds retrieval noise; docs +3, acq 30% faster | +| dense | hybrid FTS leg | 92 (ยฑ0) | โ€” | quality-neutral; hybrid is 40โ€“85% FASTER (lexical hits give the reranker cleaner candidates) โ†’ keep for latency | +| nolc | late-chunk leg | 90 (โˆ’2) | **โˆ’1 real** (4 losses, 3 gains, 9 noise) | ~1 row per 120 for 2x vectors + a second table per index โ†’ **defaulted OFF** (user decision) | +| noverify | verifier | 92 (0 flips) | โ€” | annotates, never changes answers; verdicts byte-identical on all 120 โ†’ **defaulted OFF** (user decision); latency-only cost | +| noenrich | contextual enrichment | 93 (+1) | **โˆ’1 real** (6 losses, 5 gains, 4 noise) | ~1 row per 120 for ~5x index build time (12 min vs ~55 min); kept ON for now, prime simplification candidate | + +Wall-time notes: rerank saves are modest (~10โ€“14s/query on big benches, one +batched pass). Later arms' timings are polluted by Ollama contention (the +noverify arm measured SLOWER, which is impossible) โ€” treat quality columns as +the reliable signal; timing conclusions only where the direction is +mechanical (dense-only doubling rfc wall time). + +## Actions taken (user-directed) + +- `retrieval.latechunk.enabled` โ†’ **False** in the default profile (one flag + governs both index-time build and query-time leg). +- `verification.enabled` โ†’ **False** in the default profile. +- Both remain opt-in per config/request. + +## Recommendations (not yet acted on) + +1. **Decomposition**: the +3 comes from not SPLITTING single-turn queries; + the same component also does multi-turn pronoun resolution (measured + essential โ€” multiturn.jsonl mt_07). Candidate: resolve-only mode โ€” keep the + decomposer call for context resolution, cap sub-queries at 1 unless the + query is genuinely multi-part. Needs its own A/B + multiturn gate. +2. **Reranker latency**: its function is worth 7/120. A LateOn-149M + MaxSim rescorer (Sentence Transformers v6 MultiVectorEncoder, + hf.co/blog/multi-vector-encoder) could preserve most of that at ~25x less + compute than the 4B cross-encoder. Worth an arm. +3. **Enrichment**: โˆ’1 real net for ~5x indexing cost. If indexing speed ever + matters (large corpora), turning it off is nearly free in quality. + +## Artifacts + +Session scratchpad `ablation/`: answers_{arm}.jsonl, judged4bdet_{arm}.jsonl, +panel_{A,B}_prompts.jsonl, votes_{A,B}_{1..3}.jsonl, run/judge logs, +index_noenrich_{authored,rfc}/. Baseline: arm M/fixed artifacts. diff --git a/eval/decisions/cross-bench-validation-2026-08-15.md b/eval/decisions/cross-bench-validation-2026-08-15.md new file mode 100644 index 00000000..020aa7c5 --- /dev/null +++ b/eval/decisions/cross-bench-validation-2026-08-15.md @@ -0,0 +1,54 @@ +# Cross-bench validation: are the RFC-tuned changes general? (2026-08-15) + +**Question (user-posed):** the week's five adopted changes (strict prompt, +dedupe + 12k context budget, Qwen3-Reranker-4B threshold selection, pooled +decomposition + temp-0 decompose, source-document labels) were all tuned on +the unseen-RFC bench. Do they generalize, or did we overfit to that test? + +**Method:** all 96 authored gold rows (acquisition / atlas7 / hr / docs, +24 each โ€” none touched during tuning) answered by both configurations +against IDENTICAL freshly-built product indexes (ingestion code unchanged +all week): HEAD = a3f999a, baseline = 80d5215 (post-chunker-fix, +pre-tuning) in a git worktree. qwen3.5:4b bulk judge over all 192 answers; +blind 3-voter Sonnet panels on the two direction-deciding cells (hr, docs) +for both arms. Raw rows: scratchpad authored_bench/results/. + +## Results + +| corpus | baseline | HEAD | judge | verdict | +|---|---|---|---|---| +| acq | 16/24 | 18/24 | 4b | +2, within 4b noise โ€” flat-to-positive | +| atlas7 | 19/24 | 19/24 | 4b | flat | +| hr | **24/24** | **21/24** | Sonnet panel (0 splits both) | **โˆ’3, real regression** | +| docs | 15/24 | 17/24 | Sonnet panel (0 splits both) | +2 (4 gains / 2 losses) | +| rfc (from arm history) | 5/24 | 21/24 | Sonnet panels | +16 | +| mechanical exact-substring | 24/96 | 61/96 | โ€” | +39/โˆ’2 (verbatim-copy style) | +| wall time (96 rows) | 3,928s | 2,847s | โ€” | โˆ’27% | + +## The hr regression, diagnosed + +All three lost rows (hr_h05/h08/h13) are the same shape: the HEAD answer is +CORRECT for the literal question but omits an adjacent clause the gold +includes (the 1.5x multiplier answer omits the substitute day; the +identifier/revision answer omits the effective date). Probed in-process: +the omitted text WAS in the synthesis context (the tiny hr corpus retrieves +2 chunks; both contain it; both survived selection) โ€” so the reranker/ +budget/selection stack is innocent. The strict arm-C prompt answers +narrowly; the old loose prompt padded answers with surrounding detail and +incidentally covered gold's bonus clauses. A candidate fix (a completeness +clause in the synthesis prompt: include directly-attached conditions/ +riders/dates) is NOT adopted here โ€” any prompt change must re-validate on +all five benches per this document's own lesson. + +Incidental find: late-chunk tables of tiny corpora have no FTS INVERTED +index โ€” the hybrid FTS leg fails and degrades to dense-only (the 513344f +fallback masks it). Logged as a fix-it task. + +## Verdict + +**Not overfit.** The +16 RFC gain came with flat-to-positive movement on +three of four authored benches and a โˆ’27% wall-time reduction; the single +regression (hr โˆ’3) is localized, mechanism-understood, and answer-STYLE +related (narrow literal answering), not a retrieval or grounding failure. +Changes stay adopted; the completeness-prompt experiment is queued as +future work with full 5-bench validation required. diff --git a/eval/decisions/embedder.md b/eval/decisions/embedder.md new file mode 100644 index 00000000..9ee21a18 --- /dev/null +++ b/eval/decisions/embedder.md @@ -0,0 +1,371 @@ +# Phase 1.2 โ€” embedder audit + A/B โ€” measured 2026-08-09 + +> **Superseded as a recommendation, retained as evidence.** The defaults this +> page says it did *not* change were changed at the Phase 1 adoption gate on +> 2026-08-09; what actually shipped, and the joint matrix behind it, is in +> [`../DECISIONS.md`](../DECISIONS.md). Every measurement below stands. + +Every number on this page came from a run executed on this machine on +2026-08-09. Nothing is estimated, extrapolated, or copied from a leaderboard. +Where something could not be measured, it says so. + +Raw outputs (git-ignored, re-runnable โ€” commands at the bottom): +`eval/results/p12_{qwen06b,harrier06b}_{prefix,noprefix}[_rr].json`. + +**No default was changed in `rag_system/main.py` and no user-facing doc was +updated.** This file is a recommendation, not a shipped change. + +--- + +## 1. Audit finding: the instruction prefix was **absent** + +The repo did **not** send an instruction prefix on the query side. Verified by +reading every query-embedding call site, not by grepping alone: + +| Call site | What it embedded | Prefix before this change | +|---|---|---| +| `rag_system/retrieval/retrievers.py:79` (`MultiVectorRetriever._embed_single`) | the user query, for the dense leg of hybrid search | none | +| `rag_system/agent/loop.py:310` | the raw query, for the semantic cache | none | +| `rag_system/pipelines/indexing_pipeline.py:88` | chunk text (document side) | none โ€” correct, and unchanged | + +`QwenEmbedder.create_embeddings` tokenized whatever it was handed. Queries and +documents went through the identical code path, so the query side was missing +the `Instruct: {task}\nQuery: {q}` block that both model families are trained +with. Both cards are explicit that this costs accuracy: Qwen3-Embedding claims +1โ€“5%, and harrier's card answers "Do I need to add instructions to the query?" +with "Yes, this is how the model is trained, otherwise you will see a +performance degradation." + +### What was implemented + +Query side only, in three owned files: + +* `rag_system/indexing/representations.py` โ€” `QUERY_PROMPT_TEMPLATE`, + `DEFAULT_RETRIEVAL_INSTRUCTION`, `default_query_instruction(model_name)`, + `apply_query_instruction(texts, instruction)`. `QwenEmbedder` and + `OllamaEmbedder` take a `query_instruction`; when it is non-empty that + *instance* prefixes every text it is given. `select_embedder` passes it + through as a keyword argument, defaulting to `None`. +* `rag_system/pipelines/retrieval_pipeline.py` โ€” `_query_instruction()` resolves + `config["embedding_instruction"]` โ†’ `EMBEDDING_INSTRUCTION` env var โ†’ the + model family's default; `_get_text_embedder()` passes the result. + (Edit confined to `_get_text_embedder` and the helper directly above it; the + reranker-loading section is owned by Phase 1.1 and was not touched.) +* `rag_system/indexing/embedders.py` โ€” **not modified.** It turned out to be the + LanceDB indexer, not the embedding model; nothing there needed to change. + +Family default: the official retrieval instruction for any model name matching +`qwen3-embedding` or `harrier`; empty string for everything else (bge, Ollama +tags, etc.), so non-instruction-tuned models are unaffected. + +### The document side stays unprefixed โ€” stated explicitly + +**Documents are embedded with no instruction, before and after this change, and +that is deliberate.** `IndexingPipeline` calls `select_embedder` without a +`query_instruction`, so it constructs a plain embedder; only `RetrievalPipeline` +constructs an instructed one. The asymmetry is what both model cards specify +("there is no need to add instructions to the document side") and it is also what +makes this change **index-compatible: every existing LanceDB index remains valid, +because no stored vector moves.** Only the query vector changes. + +This was verified, not assumed: turning the prefix on and off for the same +embedder reused the same cached index (the `harrier` prefix-off run completed in +18.4 s against the index the prefix-on run had built, same 313/316 chunks). + +--- + +## 2. Results + +Gold set: the committed 72-row set, unchanged. Config: `k=20`, `chunk_size=512`, +hybrid retrieval, enrichment/overviews/late-chunking/context-expansion off โ€” +i.e. exactly `BASELINE.md`'s configuration. Reranker, where used, is the shipped +`BAAI/bge-reranker-v2-m3`. + +**Control: the harness reproduced `BASELINE.md`'s first-stage numbers to the +digit** with the prefix off (docs 0.750 / 0.917 / 0.958 / 0.515; mixed 0.917 / +0.972 / 0.986 / 0.805). The instrumentation is a no-op when disabled. + +### First stage (the metric an embedder actually controls) + +| Embedder | Prefix | Corpus | R@5 | R@10 | R@20 | nDCG@10 | +|---|---|---|---|---|---|---| +| Qwen3-Embedding-0.6B | off *(= BASELINE)* | docs | 0.750 | 0.917 | 0.958 | 0.515 | +| Qwen3-Embedding-0.6B | **ON** | docs | 0.708 | 0.875 | 0.958 | 0.577 | +| harrier-oss-v1-0.6b | off | docs | 0.708 | 0.875 | 0.958 | 0.741 | +| **harrier-oss-v1-0.6b** | **ON** | docs | **0.750** | 0.875 | 0.958 | **0.759** | +| Qwen3-Embedding-0.6B | off *(= BASELINE)* | **mixed** | 0.917 | 0.972 | 0.986 | 0.805 | +| Qwen3-Embedding-0.6B | **ON** | **mixed** | 0.903 | 0.958 | 0.986 | 0.847 | +| harrier-oss-v1-0.6b | off | **mixed** | 0.903 | 0.958 | 0.986 | 0.908 | +| **harrier-oss-v1-0.6b** | **ON** | **mixed** | **0.917** | 0.958 | 0.986 | **0.915** | + +### After the `bge-reranker-v2-m3` cross-encoder + +| Embedder | Prefix | Corpus | nDCG@10 first | nDCG@10 reranked | ฮ” from reranking | +|---|---|---|---|---|---| +| Qwen3-0.6B | off | docs | 0.515 | 0.747 | **+0.232** | +| Qwen3-0.6B | ON | docs | 0.577 | 0.734 | +0.156 | +| harrier-0.6b | ON | docs | 0.759 | 0.701 | **โˆ’0.058** | +| Qwen3-0.6B | off | mixed | 0.805 | 0.908 | **+0.103** | +| Qwen3-0.6B | ON | mixed | 0.847 | 0.903 | +0.056 | +| harrier-0.6b | ON | mixed | 0.915 | 0.892 | **โˆ’0.022** | + +### Reading the tables + +**a) The prefix is not the free win the roadmap hoped for โ€” on Qwen3 it is a +trade.** On Qwen3-0.6B it buys ranking and pays in coverage: mixed nDCG@10 ++0.042, but recall@5 โˆ’0.014 and recall@10 โˆ’0.014 (one query out of 72; on docs +it is one query out of 24). Per query on docs, nDCG improved on 27 and worsened +on 14. And **the gain does not survive the cross-encoder**: post-rerank mixed +goes 0.908 โ†’ 0.903 and docs 0.747 โ†’ 0.734, i.e. slightly *worse* with the prefix +on. End to end, on the model the repo ships today, the prefix is a wash. + +**b) On harrier the prefix helps both metrics**, as its card demands: docs R@5 +0.708 โ†’ 0.750 and nDCG 0.741 โ†’ 0.759; mixed R@5 0.903 โ†’ 0.917 and nDCG 0.908 โ†’ +0.915. For harrier the prefix is not optional. + +**c) The embedder swap is the large effect.** harrier-0.6b + prefix vs the +Qwen3-0.6B baseline, first stage: **mixed nDCG@10 0.805 โ†’ 0.915 (+0.110)** and +**docs 0.515 โ†’ 0.759 (+0.244)**, at identical recall@5 (0.917 / 0.750) and +identical recall@20. On docs, 16 of 24 queries improve by >0.02 nDCG and 3 +regress. + +**d) The headline: harrier's *first stage alone* (mixed nDCG@10 0.915) beats +every reranked configuration measured here, including the shipped stack's 0.908.** +And bge-reranker-v2-m3 applied on top of harrier is **net negative** (โˆ’0.022 +mixed, โˆ’0.058 docs) โ€” it reorders a list that was already better than it is. This +is a direct input to Phase 1.1: the reranker's +0.232 on docs is largely a +first-stage-quality repair, and it shrinks or inverts as the first stage improves. +**The two decisions must be made together, not independently.** + +**e) The recall@10 regression is real and should not be waved away.** harrier +loses one query at recall@10 on mixed (0.972 โ†’ 0.958) and one on docs (0.917 โ†’ +0.875) versus the Qwen3 baseline. The specific casualty on docs is `docs_d09`, +which goes from nDCG 0.431 to a complete miss (recall@5 and @10 both 1 โ†’ 0); +`docs_d14` and `docs_d17` drop out of the top 5 but stay in the top 10. +recall@20 is identical for every arm (0.958 docs / 0.986 mixed), so nothing is +lost from the candidate set โ€” it is ordering within the top 10. + +--- + +## 3. Does harrier need code changes? No. + +`microsoft/harrier-oss-v1-0.6b` is a `Qwen3Model` (`config.json` +`model_type: qwen3`, 28 layers, hidden size 1024) with last-token pooling +(`1_Pooling/config.json`: `pooling_mode_lasttoken: true`) and L2 normalization โ€” +architecturally the same shape the existing `QwenEmbedder` path already handles. +**It loaded through that path with no model-name branch and no config entry**, on +MPS, under the repo's transformers 4.51.0 despite the card's weights being saved +by 4.57.6. + +Correctness was verified against the model card's own reference implementation +rather than assumed from "it ran": embedding the card's MS-MARCO example through +`QwenEmbedder` and through the card's `last_token_pool` + `F.normalize` snippet +gives **cosine 1.0000 between the two vectors for all four texts**, and the +score matrix matches to two decimals (65.19/25.47/30.66/70.13 vs +65.21/25.48/30.66/70.14 โ€” fp16 vs fp32). Diagonal dominance is correct. +No `NaN`/`inf`. + +Weights: 1,192,133,232 bytes, sha256 verified against the LFS blob id +`6bb124โ€ฆ8b66c`. (`huggingface_hub`'s downloader stalled repeatedly at 0 bytes on +this machine and once truncated a completed file; the weights were fetched with a +resumable `curl` and placed in the cache manually. A tooling problem, not a model +problem โ€” worth knowing before anyone else tries this download.) + +--- + +## 4. Index-rebuild implications + +| Change | Re-index required? | Measured cost | +|---|---|---| +| Turning the **query prefix** on or off | **No.** Document vectors are untouched by construction. | zero | +| Swapping **Qwen3-0.6B โ†’ harrier-0.6b** | **Yes** โ€” different vector space. | 32.2 s for the 313-chunk docs corpus, 34.9 s for 316-chunk mixed (vs 35.6 s / 37.1 s for Qwen3-0.6B). harrier is marginally *faster* to index. | +| Swapping to **Qwen3-4B** (today's shipped default) | Yes | not measured โ€” see caveats | + +**A silent-corruption hazard worth fixing in the same window.** +`VectorIndexer.index` guards an embedder swap by comparing *vector width only* +(`embedders.py`, `_table_vector_dim`). harrier-0.6b and Qwen3-Embedding-0.6B are +**both 1024-dim**, so swapping between those two would pass the guard and append +mutually-unintelligible vectors to an existing table with no error. The shipped +4B default is 2560-dim, so a 4B โ†’ harrier swap *would* be caught โ€” but the +0.6B โ†’ harrier path would not. Recommendation: stamp the embedder model name into +the table (or a sidecar) and compare it, alongside the dim check, during the +re-index window. Not done here: it is an index-format change, and this file's +mandate was to measure, not to alter the index format. + +The eval harness itself is already safe on two independent axes, verified: +`eval/run_eval.py` writes each embedder's index to +`eval/.eval_indexes/<slug(embedder)>/<corpus>/` **and** carries `embedder` inside +the fingerprint dict, so two embedders can neither share a directory nor accept +each other's cached marker. **No fingerprint fix was needed and `run_eval.py` was +left unmodified.** + +--- + +## 5. Licences + +| Model | Licence | Verdict | +|---|---|---| +| `Qwen/Qwen3-Embedding-0.6B` / `-4B` | Apache 2.0 | fine | +| `microsoft/harrier-oss-v1-0.6b` | **MIT** (confirmed in the model card's front-matter: `license: mit`) | fine โ€” MIT is strictly more permissive | + +Neither licence blocks anything. Both allow commercial use and redistribution. + +--- + +## 6. Latency + +The GPU was shared with another agent's reranker A/B throughout, so **latency +here is noisy and the quality numbers are not.** First-stage means per query +across all arms: 100โ€“168 ms, with harrier (103โ€“120 ms) indistinguishable from +Qwen3-0.6B (100โ€“109 ms) โ€” same architecture, same 1024-dim output, one extra +short prefix to tokenize. Rerank means ranged 1884โ€“3308 ms for the identical +20-candidate cross-encoder workload, against `BASELINE.md`'s 1741โ€“1788 ms; that +spread is contention, not signal. **Do not quote the rerank column as a +measurement.** + +Determinism was checked rather than assumed: re-running one identical +configuration end to end produced bit-identical metrics +(docs R@5 0.7500, R@10 0.9167, nDCG first 0.5150, nDCG reranked 0.7468 twice). + +--- + +## 7. Recommendation + +1. **Adopt `microsoft/harrier-oss-v1-0.6b` as the embedder**, with the query-side + instruction prefix on. On the only corpus with real distractors it is + +0.110 nDCG@10 first-stage over the Qwen3-0.6B baseline at equal recall@5 and + equal recall@20, it needs no code beyond what is already merged, it indexes + slightly faster, and MIT is the more permissive licence. **Gate: this must land + in the same re-index window as the Phase 1.1 reranker decision**, because + finding (d) says bge-reranker-v2-m3 is net-negative on top of harrier โ€” adopting + the embedder without revisiting the reranker would ship a stack whose last + stage costs ~2 s per query to make the ranking *worse*. +2. **Keep the prefix on for harrier; treat it as undecided for Qwen3.** If the + repo stays on a Qwen3 embedder, the honest reading of the measurement is that + the prefix is a wash-to-slightly-negative end to end, and + `embedding_instruction: ""` should be set explicitly rather than inheriting the + family default. +3. **Do not change `main.py` yet.** Both of the above are recommendations + pending the Phase 1.1 outcome and the 4B measurement below. + +--- + +## 8. Caveats โ€” what this does not tell you + +* **Nothing about the shipped default.** The comparison is against + Qwen3-Embedding-**0.6B**; the repo ships Qwen3-Embedding-**4B**. The 4B was not + measured: it is ~8 GB and measured throughput to the HF CDN on this machine was + **630โ€“670 KB/s**, i.e. roughly 3.5 hours of download, on a machine already + shared with another agent's GPU work. A 4B-vs-harrier-0.6b number is the single + most valuable missing data point and should be the first thing run when + bandwidth allows. It is entirely possible the 4B closes some of the +0.110 gap. +* **The prefix now defaults ON for Qwen3-Embedding-4B, untested.** The + family-default rule matches `qwen3-embedding`, so the shipped 4B default now + receives a prefix nobody has measured. Given that the prefix measured as a wash + on the 0.6B, this is a live risk, not a certainty of improvement. One-line + revert: set `embedding_instruction: ""` in the profile. +* **Post-rerank numbers are internally comparable only.** All arms here ran within + the same hour against the same tree, and a repeat was bit-identical โ€” but they + do **not** reproduce `BASELINE.md`'s post-rerank column (docs 0.747 here vs + 0.731 there). Another agent is actively editing `rag_system/rerankers/reranker.py` + and `retrieval_pipeline.py`'s reranker section for Phase 1.1, so the tree changed + between the two measurements. First-stage numbers **do** reproduce BASELINE + exactly, which is why the recommendation rests on them. +* **72 queries.** A 0.014 recall delta is one query. Deltas below ~0.03 on recall + should not be treated as findings; the nDCG gaps driving the recommendation + (+0.110, +0.244) are an order of magnitude larger than that floor. +* **Vectors are never L2-normalized and the vector search uses LanceDB's default + L2 metric** (`retrievers.py` calls `tbl.search(vector).limit(k)` with no + `.metric()`), while both model cards specify cosine similarity over normalized + embeddings. Every arm above is affected identically, so the comparison is fair, + but all of them may be leaving accuracy on the table. Not changed here: + normalizing is a document-side change that invalidates existing indexes, and + `retrievers.py` is outside this task's ownership. **Worth testing in the + re-index window** โ€” it is the cheapest remaining lever. +* **The semantic cache is affected but harmlessly.** It compares query-to-query, + so both sides carry the prefix. Measured on the 72 gold queries, the prefix + *lowers* mean pairwise cosine (0.361 โ†’ 0.216) and max (0.928 โ†’ 0.899); with the + shipped `semantic_cache_threshold` of 0.98 nothing crosses the bar either way, + so the cache only ever fires on near-identical queries, as before. +* **Nothing about answer quality.** These are retrieval metrics; synthesis, + verification and groundedness are untouched by this file. +* `atlas7` and `hr` are 1 and 2 chunks and saturate at recall 1.0 by construction. + They are omitted above for that reason โ€” read `mixed`. + +--- + +## 9. Reproducing every number above + +```bash +cd /path/to/localGPT + +# control: reproduces BASELINE.md's first-stage numbers exactly +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B EMBEDDING_INSTRUCTION="" \ + .venv/bin/python eval/run_eval.py --corpus all --no-rerank \ + --json-out eval/results/p12_qwen06b_noprefix.json + +# Qwen3 + query prefix (family default; omit EMBEDDING_INSTRUCTION) +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ + .venv/bin/python eval/run_eval.py --corpus all --no-rerank \ + --json-out eval/results/p12_qwen06b_prefix.json + +# harrier, prefix off / on +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b EMBEDDING_INSTRUCTION="" \ + .venv/bin/python eval/run_eval.py --corpus all --no-rerank \ + --json-out eval/results/p12_harrier06b_noprefix.json +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b \ + .venv/bin/python eval/run_eval.py --corpus all --no-rerank \ + --json-out eval/results/p12_harrier06b_prefix.json + +# add --reranker BAAI/bge-reranker-v2-m3 and drop --no-rerank for the +# post-rerank column, e.g. +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b \ + .venv/bin/python eval/run_eval.py --corpus all \ + --reranker BAAI/bge-reranker-v2-m3 \ + --json-out eval/results/p12_harrier06b_prefix_rr.json +``` + +`EMBEDDING_INSTRUCTION=""` switches the prefix off; unset inherits the model +family's default; any other string overrides the task description. The same +knob exists as `config["embedding_instruction"]`, which takes precedence. + +--- + +## Gate validation (2026-08-09, run by the validating orchestrator, not the Phase-1.2 agent) + +Independent reproduction: harrier-0.6b + prefix on `mixed` reproduced the agent's +headline exactly (R@5 0.917 / R@10 0.958 / nDCG@10 first-stage **0.915**; +`eval/results/` + `/tmp/val_harrier.json`). + +The declared 4B gap is now closed. Qwen/Qwen3-Embedding-4B (the shipped default), +first stage, both prefix modes โ€” commands as in ยง9 with the 4B model id: + +| Config | mixed R@5 | mixed R@10 | mixed R@20 | mixed nDCG@10 | docs nDCG@10 | 1st-stage ms | +|---|---|---|---|---|---|---| +| 4B + prefix (family default) | 0.889 | 0.931 | 0.972 | 0.875 | 0.638 | ~334 | +| 4B, prefix off | 0.889 | 0.931 | 0.986 | 0.816 | 0.518 | ~366 | +| harrier-0.6b + prefix (ref) | 0.917 | 0.958 | 0.986 | **0.915** | **0.759** | ~133 | + +**Finding: the 8 GB shipped default is dominated by the 1.2 GB harrier on this +gold set in both prefix configurations โ€” on ranking quality, recall@5/@10, +latency (~3x), and memory (~7x).** The prefix helps 4B's ranking (+0.059 nDCG) +while leaving recall unchanged, so if 4B is retained anywhere, keep the prefix. + +Honest scope limits: 72 English queries, one machine, digital-born corpora. +Qwen3-4B's documented advantages (multilingual breadth, MRL dimension +flexibility, 32K context) are not exercised by this gold set, so the right +docs framing is "harrier-0.6b default, Qwen3-Embedding-4B documented option +for multilingual/long-context corpora" โ€” not "4B is bad". + +Adoption remains gated on the joint reranker decision (Phase 1.1) plus the two +hazards above (width-only index guard; L2-vs-cosine normalization), all to land +in one re-index window. + + +--- + +**Gate correction (2026-08-09, post-adoption):** ยง8's statement "Vectors are never +L2-normalized and the vector search uses LanceDB's default L2 metric" described the +tree at measurement time and is now false: v4 tables L2-normalize at write and query +(`rag_system/indexing/embedders.py`), making L2 ordering the cosine ordering. The ยง8 +fairness argument (all arms measured under identical metric handling) still holds. diff --git a/eval/decisions/experiments-resolveonly-maxsim-2026-08-19.md b/eval/decisions/experiments-resolveonly-maxsim-2026-08-19.md new file mode 100644 index 00000000..da38de39 --- /dev/null +++ b/eval/decisions/experiments-resolveonly-maxsim-2026-08-19.md @@ -0,0 +1,88 @@ +# Experiments: resolve-only decomposition + LateOn MaxSim reranker โ€” 2026-08-19 + +Both user-approved follow-ups from the component ablation +(component-ablation-2026-08-18.md). **Verdicts: neither is adopted.** The +runs also surfaced a multi-turn cost of the latechunk-off default and a +2-row multi-turn regression from the fix-set window โ€” both documented below. + +## 1. Resolve-only decomposition โ€” NOT ADOPTED + +Hypothesis: keep the decomposer's context resolution (multi-turn needs it), +stop splitting. Implementation: `query_decomposition.resolve_only` flag โ€” +same LLM call, same frozen prompts (single-turn dump byte-identity +unaffected), pipeline uses `resolved_query` instead of the splits. + +- Single-turn (120 rows, 4b-det): resolveonly 90 vs its proper control + (nolc, same latechunk-off defaults) 90 โ€” **neutral**. The nodecomp arm's + +3 came from using the RAW query; the decomposer's rewriting of + self-contained questions costs the gains back. +- Multi-turn: 9/12, but the control (same defaults, resolve_only off) also + scored 9/12 with identical misses โ€” resolve-only exonerated there. +- Verdict: no benefit anywhere; flag stays in the code, default False, as a + documented negative result. The single-turn win the ablation pointed at + requires bypassing the decomposer entirely, which multi-turn cannot afford + as a global default. A triage-style "history-empty โ†’ skip decomposer" + fast path remains the plausible shape; not attempted here. + +## 2. LateOn-149M MaxSim rescorer โ€” NOT ADOPTED as default; kept as opt-in + +Sentence Transformers v6 `MultiVectorEncoder` (hf.co/blog/multi-vector- +encoder). ST6 requires transformers 5.x / torch โ‰ฅ2.5 โ€” incompatible with the +repo's pinned MPS stack (an in-place install broke the venv and was rolled +back to transformers==4.51.0) โ€” so the model runs OUT OF PROCESS in its own +venv behind a localhost scoring endpoint; `reranker.strategy: "maxsim"` +selects the thin client (`MaxSimRerankerScorer`). + +- Quality vs the Qwen-4B control (nolc, same defaults): 90 โ†’ 87 (4b-det). + Sonnet panels on all 13 down-flips: **11 REAL losses, 0 split votes** โ€” + including the rfc attribution/crossref rows (q06, q18, q20, q21). Not churn. +- Latency: the rerank stage drops from ~25โ€“30s to ~1โ€“2s; end-to-end docs + 62โ†’18s, hr 22โ†’9s, atlas7 14โ†’8s, rfc 58โ†’40s per query. +- Known confound: maxsim ran fixed top-10 selection (MaxSim scores are not + calibrated probabilities, so the min_score threshold โ€” gated on + isinstance(QwenRerankerScorer) โ€” does not apply). Some flips may be + selection-policy, not scorer quality; untangling would need a + qwen-top10-no-threshold control arm. +- Verdict: real quality cost (~3+ net real rows/120) for a large speed win. + Not the default. Kept as an experimental opt-in for latency-sensitive + setups; sidecar setup documented in scratch `maxsim/server.py` (not + production-packaged). + +## 3. Incidental finding: multi-turn attribution matrix + +The multiturn set (12 conversations) was re-run under four configs on +current code: + +| config | score | misses | +|---|---|---| +| defaults (lc off, verify off) | 9/12 | mt_03, mt_11, mt_12 | +| + resolve_only | 9/12 | same three | +| lc ON, verify off | 10/12 ร—2 runs | mt_11, mt_12 | +| lc ON, verify ON (m1e-era config) | 10/12 | mt_11, mt_12 | + +- **mt_03 is a real latechunk casualty**: the latechunk-off default costs + this conversation (follow-up phrasing drifts from document wording; the + document-context vectors were finding it). Single-turn ablation showed + โˆ’1/120; multi-turn adds this cost the ablation gate could not see. +- **mt_11/mt_12 (both rfc) fail under EVERY current config** including the + exact config that scored 12/12 three times on 2026-08-16 โ€” the regression + entered with the code changes between m1e and now (fix-set 3a8ebd9 and + after). Prime suspect: the retry-judge temperature-0 pin changing + keep/reject decisions on borderline rfc retrievals. Both rows are + answer-entity conversations whose turn-2 retrieval is marginal. Queued + for diagnosis; n=12 diagnostic set, 2 rows โ‰ˆ its noise floor, but the + consistency across 4 runs makes it real. + +## Standing recommendation for the user + +Re-enabling latechunk recovers mt_03 (+1/12 multi-turn) and the โˆ’1/120 +single-turn panel cost of removing it โ€” at the price of 2x vectors and a +second table per index. The verifier flip remains cost-free for quality. +The mt_11/12 diagnosis is queued follow-up work either way. + +## Artifacts + +Scratch `ablation/`: answers/judged4bdet for resolveonly + maxsim, +panel_ms_prompts.jsonl, votes_ms_{1..3}.jsonl. Scratch `multiturn/`: +mt_answers_{mt_resolveonly,mt_ctrl_newdefaults,mt_lcon,mt_lcon2,mt_fullcfg}.jsonl. +Scratch `maxsim/`: sidecar venv + server.py + server.log. diff --git a/eval/decisions/fixset-impact-2026-08-17.md b/eval/decisions/fixset-impact-2026-08-17.md new file mode 100644 index 00000000..bb74d98a --- /dev/null +++ b/eval/decisions/fixset-impact-2026-08-17.md @@ -0,0 +1,77 @@ +# Code-review fix-set: measured impact (arm M / "fixed") โ€” 2026-08-17 + +**Verdict: fix-set stays.** Net quality across the five benches is +flat-to-positive with one confirmed trade on rfc. All changes are +correctness fixes, not tuning; the indexes every prior number was measured +on are now known to have been structurally degraded. + +## What was measured + +Commits `3a8ebd9`+`3f78be4` changed index *content* (docling walk restored +heading paths + reading order; latechunk got real vectors past 8192 tokens +instead of identical CLS-garbage; split_markdown loop fixed; enrichment +windows stopped crossing document boundaries; FTS on `_lc` from build). So +all 5 bench indexes were REBUILT with the new code (v2 dirs; v1 kept), and +the full 120-row E2E suite + 12-conversation multiturn set re-ran on them. +Baseline: arm K answers (pre-fix code, v1 indexes) and the arm-I Sonnet +panel record. + +Index deltas alone tell the story of bug 2.6: acq 13โ†’82 chunks, hr 2โ†’10, +atlas7 1โ†’5, docs 366โ†’608 (the old walk collapsed structure); rfc 683โ†’683 +(txt path โ€” its change is enrichment + latechunk vectors, not chunking). + +## Results + +| bench | pre-fix | post-fix | judge | +|---|---|---|---| +| acq | 19/24 | 19/24 | deterministic 4b (temp 0) | +| atlas7 | 22/24 | 23/24 | deterministic 4b | +| hr | 19/24 | 21/24 | deterministic 4b | +| docs | 14/24 | 15/24 | deterministic 4b | +| authored total | 74/96 | **78/96** | deterministic 4b | +| rfc (full Sonnet panel) | 21/24 (arm I record) | **19/24** | 3-voter Sonnet, 1 split/72 | +| multiturn | 12/12 | **12/12** | mechanical | +| rfc mean wall | 59s | 51s (โˆ’14%) | โ€” | + +Sonnet arbitration of every 4b down-flip (12 rows ร— both arms, 3 voters): +7 of 12 were 4b noise; **5 real losses** โ€” acq_q15, docs_d18, docs_d20, +rfc_q06, rfc_q24. Since the authored total still rose +4 *including* its 3 +real losses, the authored gains are real and larger. + +## The notable individual outcomes + +- **hr_h05 RECOVERED (3/3)** โ€” the "2 days extended family" row that the + completeness-prompt clause could not fix (arm J, reverted). With hr now + chunked 2โ†’10, retrieval surfaces the bereavement section as its own + chunk and the model states both durations. Confirms the arm-J diagnosis: + it was an attention/structure problem, never an instruction problem. + hr_h08/h13 remain failed (0/3) โ€” same qualifier-omission style issue. +- **rfc q10, q15, q20 now PASS** โ€” q10/q15 were the long-standing + single-doc residue of arm I. The real latechunk vectors changed which + candidates the lc leg contributes on long RFCs. +- **rfc q06, q24 now FAIL (real, panel-confirmed)** โ€” previously-passing + crossref rows lost by the same candidate redistribution. q04/q11/q17 + also fail at M, but q04 was already failing at arm K (pre-fix โ€” panel + batch 1) and q17 was arm I residue; they are not fix-set losses. + +## Interpretation + +rfc 21โ†’19 is a โˆ’2 net from a changed retrieval surface: two real losses, +two-to-three real gains elsewhere in the same bench, on n=24 where the +documented noise floor is 1โ€“2 rows. The losses were bought by removing +objectively-wrong behavior (identical garbage vectors indexed as real +data). Reverting correctness fixes to protect two bench rows would be +tuning-to-the-bench in reverse. The fix-set stays; q06/q24 join the +residue list as diagnosable candidates (both crossref rows โ€” the crossref +residue remains the top rfc lever, item 1.8). + +## Artifacts + +`fiximpact/` in the session scratchpad: build_v2.log, e2e logs, +`authored_e2e_answers_fixed.jsonl`, `rfc_e2e_answers_m.jsonl`, +`mt_answers_m1e.jsonl`, `judged4bdet_{fixed,rfc_m}.jsonl`, +`panel{,2}_prompts.jsonl`, `votes_{1..3}.jsonl`, `votes2_{1..3}.jsonl`. +v2 indexes: `{authored_bench,rfc_shakedown}/product_index_v2/`. +Note: the fix-set session deleted the scratch rfc E2E runner; it was +reconstructed as `fiximpact/run_rfc_m2.py` โ€” committing a repo-adapted rfc +runner (mirroring eval/multiturn/) remains open. diff --git a/eval/decisions/ftslc-index-fix-2026-08-15.md b/eval/decisions/ftslc-index-fix-2026-08-15.md new file mode 100644 index 00000000..b5922365 --- /dev/null +++ b/eval/decisions/ftslc-index-fix-2026-08-15.md @@ -0,0 +1,57 @@ +# FTS index on late-chunk tables (arm K) โ€” 2026-08-15 + +**Verdict: ADOPTED.** Bug fix restoring designed hybrid retrieval on the +late-chunk leg; measured E2E-neutral on all 5 benches, Sonnet-panel-confirmed +zero regressions. Plus one methodological finding: the 4b screen judge was +nondeterministic; now pinned to temperature 0. + +## The bug + +`IndexingPipeline` created an FTS index only on the base text table; the +late-chunk block indexed vectors into `<table>_lc` and never created one. +Consequence: **every** `_lc` table in existence lacked an FTS index โ€” verified +across all scratch indexes, including the 683-row rfc table โ€” so the hybrid +retriever's FTS leg failed on `_lc` and silently degraded to dense-only +(retrievers.py graceful-degradation path, added 2026-08-12, masked it). The +chip (task_c7eaedfb) guessed "tiny corpora"; the truth is the lc leg has been +dense-only everywhere since late-chunking landed. + +## The fix + +- `indexing_pipeline.py`: after the late-chunk indexing loop, create + `text_idx` on the lc table (same guard/naming as the base-table block). +- Existing tables need no re-ingest: `create_fts_index` on the 5 scratch + `_lc` tables in place; smoke-verified the FTS leg now returns rows on + `hr_product_v1_lc` and "FTS leg failed" no longer appears in arm-K logs. + +## Measurement (arm K: identical code, index present vs absent) + +Full 120-row rerun vs the arm-I/HEAD answers (0 errors): + +- Mechanical: substring rfc 10โ†’9, docs 17โ†’16, others flat (noise floor); + cited-documents changed on only **5/96** authored rows; many answers + byte-identical (hr 20/24, atlas7 19/24); rfc mean wall 74sโ†’59s. +- Expected, and explains the flatness: lc rows carry the same text as base + rows, so the restored lexical leg mostly re-finds what base FTS already + found and RRF fusion barely moves. +- Sonnet panel (3 blind voters, 18 cells = 9 genuinely-differing down-flipped + rows ร— both arms): **every row judges identically across arms** (8/9 pass + both, docs_d17 fail both), 2 split votes, 0 cell flips. Zero regressions. + +## Judge-noise finding (4b screen) + +The 4b coarse screen initially reported authored 69โ†’59 โ€” but 11 of its 24 +verdict flips (including all four hr flips) were on **byte-identical +answers**: the judge itself sampled at default temperature. `eval/judge.py` +now passes `options={"temperature": 0}` on the Ollama path. Prior 4b screen +numbers remain valid only as coarse ordering, never row-level evidence โ€” +which is already their documented status. + +## Status + +- Fix + judge determinism committed. Rebuild is NOT required for existing + user indexes, but `create_fts_index("text", use_tantivy=False)` on `_lc` + tables is the one-liner migration; new indexes get it automatically. +- Artifacts: `rfc_e2e_answers_k.jsonl`, `authored_e2e_answers_ftslc.jsonl`, + `judged4b_{ftslc,rfc_k}.jsonl`, `ftslc_panel_prompts.jsonl`, + `votes_ftslc_{1,2,3}.jsonl` in the session scratchpad. diff --git a/eval/decisions/glm-ocr-spike.md b/eval/decisions/glm-ocr-spike.md new file mode 100644 index 00000000..2a8652a0 --- /dev/null +++ b/eval/decisions/glm-ocr-spike.md @@ -0,0 +1,469 @@ +# GLM-OCR feasibility spike โ€” Apple Silicon (roadmap Phase 1.3) + +_Run 2026-08-08 on the machine below. **Investigation only โ€” no pipeline code was +changed.** Every claim here is either a command run on this host (with its output) +or a cited URL. Anything I could not verify is labelled **unverified**._ + +**Recommendation: GO-LATER** โ€” the serving path works here today with zero new +Python dependencies, but three defects (below) block making it a default. See +[Recommendation](#recommendation). + +| | | +|---|---| +| Host | Apple M2 Max, 96 GiB unified memory, macOS 15.5 (24F74) | +| Ollama | 0.32.6 (`ollama --version`) | +| Python env | `.venv` โ€” docling 2.118.1, docling-core 2.91.0, docling-parse 7.11.0, torch 2.4.1, transformers 4.51.0, pymupdf 1.28.2 | +| Model pulled | `glm-ocr:latest`, id `6effedd0dc8a`, 2.2 GB โ€” **kept** (it works) | + +--- + +## Q1 โ€” Serving options on Apple Silicon, Aug 2026 + +### (a) Ollama โ€” WORKS, this is the viable path + +`glm-ocr` is in the **official Ollama library namespace** (no user prefix): +<https://ollama.com/library/glm-ocr>. Tags on that page: `latest` 2.2 GB, +`q8_0` 1.6 GB, `bf16` 2.2 GB โ€” all 128K context, text+image input. The upstream +repo links its own Ollama guide +(<https://github.com/zai-org/GLM-OCR/blob/main/examples/ollama-deploy/README.md>), +so the tag is vendor-blessed, not a random community upload. + +``` +$ ollama pull glm-ocr # ~17 min here at ~2.0 MB/s; "success" +$ ollama list | grep ocr +glm-ocr:latest 6effedd0dc8a 2.2 GB +``` + +``` +$ ollama show glm-ocr + architecture glmocr + parameters 1.1B + context length 131072 + embedding length 1536 + quantization F16 + requires 0.15.5 + Capabilities: completion, vision, tools +``` + +Loaded footprint, from `ollama ps` during a run: **2.8 GB, 100% GPU** at +`num_ctx=16384`. Disk: 2.1 GB. + +**Caveat found by experiment โ€” the Ollama build ignores the prompt.** + +``` +$ ollama show glm-ocr --template +{{ .Prompt }} +$ ollama show glm-ocr --parameters +temperature 0 +``` + +The Modelfile template is bare and defines no stop strings. I sent three +different prompts (`Text Recognition:`, a hand-wrapped +`<|user|>\nText Recognition:<|assistant|>`, and `Convert this page to markdown.`) +against the same page: **byte-identical output all three times**. Consequence: +GLM-OCR's documented alternate modes (`Formula Recognition:`, +`Table Recognition:`, and the JSON information-extraction schemas) are **not +reachable through Ollama**. Only the default full-page recognition behaviour is. + +Upstream's own Ollama README says to prefer the native `/api/generate` endpoint +over the OpenAI-compatible one for vision, and recommends vLLM/SGLang for +production. Docling uses `/v1/chat/completions` โ€” it worked here regardless +(Q2), but that is a documented mismatch to keep an eye on. + +### (b) llama.cpp / GGUF โ€” supported upstream, not tested here + +Support landed via PR #19677; usage thread: +<https://github.com/ggml-org/llama.cpp/discussions/19721>. Needs the decoder +GGUF **plus** an `mmproj` projector from `ggml-org/GLM-OCR-GGUF`, and +**flash-attention must be off**: + +``` +llama-server -m glmocr-Q4_K_M.gguf --mmproj mmproj-glmocr-Q4_1.gguf \ + -c 12000 -ngl 99 --flash-attn off -fit off +``` + +Reported working on M-series Macs in that thread (one user: ~3 min for a 7-page +document on an M1 Air at Q8). **Not exercised on this host** โ€” Ollama (which is +llama.cpp underneath) already gave us a working server, so a second one buys +nothing except control over the chat template, which is the one thing that would +fix the prompt-ignored caveat above. Worth revisiting only if we need the +`Table Recognition:` / extraction modes. + +### (c) MLX โ€” a port exists; needs a dependency we did not install + +- `mlx-community/GLM-OCR-bf16`, 2.21 GB, BF16, MIT + (<https://huggingface.co/mlx-community/GLM-OCR-bf16>). Card states it was + converted with **mlx-vlm 0.3.11**. +- Our pinned docling already knows about it โ€” `vlm_model_specs.py:436` + (`GLMOCR_MLX`, `repo_id="mlx-community/GLM-OCR-bf16"`, MPS-only) and + `stage_model_specs.py:1367-1368` ("Native GLM-OCR support was added to + mlx-vlm in v0.3.11"). +- Neither `mlx` nor `mlx-vlm` is in `.venv`, and installing packages was out of + scope for this spike, so **this path is unverified on this host**. It is the + most promising *future* path: in-process, no server, full prompt control, + and docling drives it through the same preset with + `engine_options=MlxVlmEngineOptions(...)`. + +### (d) vLLM on macOS โ€” NO + +<https://docs.vllm.ai/en/stable/getting_started/installation/cpu/> and +`docs/getting_started/installation/cpu/apple.inc.md` in the vLLM repo: macOS is +**source-build, CPU-only, FP32/FP16, no prebuilt Apple Silicon wheels**. A 2026 +write-up measures the CPU backend at 20โ€“30ร— slower than llama.cpp's Metal +backend. A community Metal plugin exists +(<https://github.com/vllm-project/vllm-metal>, MLX compute backend) but that is +a third serving stack to own. Confirmed dead end for us; not pursued further. + +--- + +## Q2 โ€” Docling integration + +### Our pinned version already ships it. No upgrade needed. + +``` +$ .venv/bin/pip show docling +Name: docling +Version: 2.118.1 +``` + +GLM-OCR is present in that installed tree: + +| File (in `.venv/lib/python3.12/site-packages/docling/`) | What | +|---|---| +| `datamodel/vlm_model_specs.py:416` | `GLMOCR_TRANSFORMERS` (`zai-org/GLM-OCR`, MPS listed as a supported device) | +| `datamodel/vlm_model_specs.py:436` | `GLMOCR_MLX` (`mlx-community/GLM-OCR-bf16`) | +| `datamodel/vlm_model_specs.py:446/449` | `GLMOCR_VLLM`, `GLMOCR_VLLM_API` | +| `datamodel/stage_model_specs.py:1354` | preset `glm_ocr` with `api_overrides` for `API`, `API_OPENAI`, **`API_OLLAMA`**, `API_LMSTUDIO` | +| `models/inference_engines/vlm/base.py:29` | `VlmEngineType` incl. `API_OLLAMA` | + +Verified by instantiating it (no network, no code change): + +``` +$ .venv/bin/python -c " +from docling.datamodel.pipeline_options import VlmConvertOptions +from docling.datamodel.vlm_engine_options import ApiVlmEngineOptions, VlmEngineType +o = VlmConvertOptions.from_preset('glm_ocr', + engine_options=ApiVlmEngineOptions(engine_type=VlmEngineType.API_OLLAMA)) +print(o.engine_options.url, o.model_spec.api_overrides[VlmEngineType.API_OLLAMA].params)" +http://localhost:11434/v1/chat/completions {'model': 'glm-ocr', 'max_tokens': 4096} +``` + +Preset defaults: prompt `Text Recognition:`, `scale=2.0`, `max_tokens=4096`, +`temperature=0.0`, `response_format=MARKDOWN`, `stop_strings=['<|user|>','<|endoftext|>']`. + +**So the roadmap's "requires Docling โ‰ฅ v2.84" is satisfied โ€” and then some. The +upgrade question is moot; there is nothing to upgrade and nothing to break.** +(I could not read the docling CHANGELOG to pin the exact version that first +added the preset โ€” GitHub returned a render error for `CHANGELOG.md`. It does +not matter for the decision, but it is **unverified** which release introduced it.) + +### End-to-end, on this host, with the pinned version + +```python +# exactly what was run (read-only script, nothing written into the repo) +vlm = VlmConvertOptions.from_preset("glm_ocr", + engine_options=ApiVlmEngineOptions(engine_type=VlmEngineType.API_OLLAMA)) +opts = VlmPipelineOptions(vlm_options=vlm, enable_remote_services=True) +conv = DocumentConverter(format_options={ + InputFormat.PDF: PdfFormatOption(pipeline_cls=VlmPipeline, pipeline_options=opts)}) +conv.convert(pdf) +``` + +Two gotchas, both hit for real: + +1. **`enable_remote_services=True` is mandatory even for `localhost`.** Without + it: `docling.exceptions.OperationNotAllowed: Connections to remote services + is only allowed when set explicitly.` If we ever ship this, that flag has to + be documented โ€” it *sounds* like it sends documents off-box and it does not. +2. Harmless shutdown noise: `Exception ignored in VlmConvertModel.__del__ โ€ฆ + 'NoneType' object has no attribute 'warning'` (docling bug, at exit only). + +### Alternative worth knowing about (not tested) + +`DCC-BS/docling-glm-ocr` (<https://github.com/DCC-BS/docling-glm-ocr>, +PyPI `docling-glm-ocr`) plugs GLM-OCR in as an **`ocr_options` engine inside the +classic PDF pipeline** โ€” layout/table models keep running, and GLM-OCR only +recognises the OCR regions docling asks for. That is architecturally closer to +what localGPT wants (replace the OCR box, keep the rest) than `VlmPipeline`, +which replaces the whole pipeline. It targets a vLLM endpoint. **Unverified โ€” I +did not install it.** Flagging it because it is the natural answer to defect #2 +below. + +--- + +## Q3 โ€” End-to-end parse test + +Method: render the page with PyMuPDF at the preset's `scale=2.0`, POST as a +base64 PNG. Scripts live in the scratchpad (`glmocr_smoke.py` โ†’ OpenAI-compatible +endpoint, `glmocr_native.py` โ†’ `/api/chat` with `num_ctx`), not in the repo. + +### `eval/corpora/atlas7_service_manual.pdf` page 1 โ€” verbatim model output + +``` +Atlas-7 Espresso Machine ยท Service Manual + +Model: Atlas-7 Dual Boiler (2026 revision C) +Manufacturer: Meridian Coffee Systems, Tacoma WA + +1. OPERATING SPECIFICATIONS +The brew boiler operates at a pressure of 9.2 bar during extraction. +The steam boiler is maintained at 1.45 bar. The PID controller keeps brew water at 93.5 degrees Celsius with a tolerance of 0.4 degrees. +The vibratory pump is rated for 52 watts continuous duty. + +2. MAINTENANCE SCHEDULE +Descaling must be performed every 60 days when water hardness exceeds 120 ppm. The group head gasket (part MG-311) should be replaced every 14 months. Backflushing with Cafiza detergent is recommended weekly. +``` + +Against the PDF's own text layer: **every planted fact is exact** โ€” 9.2 bar, +1.45 bar, 93.5 ยฐC, 0.4 ยฐC, 52 W, 60 days, 120 ppm, `MG-311`, 14 months, "Cafiza". +No hallucinated content. Only difference from the source is that hard line +wraps are joined into paragraphs, which is what we want for chunking. +Page 2 likewise exact (`TS-71`, `E11/E23/E42/E57`, 12 bar, 200 ml, 36-month, +8 percent, "under the drip tray on the left rail"). + +### Defect #1 โ€” the page is sometimes transcribed twice + +Deterministic, page-dependent, and it survives everything I threw at it: + +``` +atlas7 p1: 5.48s chars=1284 half-vs-half similarity=0.97 DUPLICATED +atlas7 p2: 2.16s chars=1282 half-vs-half similarity=0.97 DUPLICATED +northwind p1: 4.39s chars=846 similarity=0.05 ok +northwind p2: 4.27s chars=737 similarity=0.06 ok +northwind p3: 4.37s chars=835 similarity=0.08 ok +invoice_scan p1: 3.68s chars=1337 similarity=0.81 DUPLICATED +``` + +The second copy is usually wrapped in a ```` ```markdown ```` fence, and on +atlas7 p2 it actually contains a running header the *first* copy dropped. Ruled +out: `stop` strings (`<|user|>`, `<|endoftext|>` โ€” passed explicitly, no effect), +`num_ctx` (4096 vs 16384 โ€” identical output, so it is not context shift), +prompt wording (ignored entirely, see Q1a). Cause is the bare Ollama template / +missing stop token, i.e. it is a **packaging** problem, not a model problem โ€” +which is consistent with the llama.cpp thread where a clean install fixed +repeat artifacts for another user. Cheap mitigation: de-duplicate identical +halves in post-processing. Real fix: MLX or a self-hosted llama-server with the +correct chat template. + +### Scanned + tabular test vs. what localGPT runs today + +Fixture (scratch, generated for this spike): a bordered 6ร—5 parts-invoice table +plus prose, rasterised at 150 dpi, greyscaled, rotated 0.6ยฐ, JPEG q45 โ€” +**0 characters of text layer**, so it takes the OCR branch. + +**GLM-OCR (12.6 s cold / 3.7 s warm):** + +``` +| Part No. | Description | Torque (Nm) | Qty | Price (EUR) | +| MG-311 | Group head gasket, 8.5 mm | 12.5 | 2 | 4.20 | +| TS-71 | Brew thermistor, PT1000 | 3.0 | 1 | 18.75 | +| OPV-12 | Over-pressure valve, 12 bar | 22.0 | 1 | 31.40 | +| FM-9 | Flow meter, 0.45 ml/pulse | 6.5 | 1 | 27.90 | +| PMP-52 | Vibratory pump, 52 W | n/a | 1 | 63.05 | +Subtotal 145.30 VAT 19% 27.61 Total 172.91 +``` + +**All 30 cells correct.** + +**Current chain (`rag_system.ingestion.document_converter`, unmodified, 6.7 s):** + +``` +| | | CCC CCE | | +| oa | [ompmergmennsem | _ CE | CES | +| | Brew thermistor, PT1000 | 3.0 | a | | +| | Over-pressure valve, 12 bar | 22.0 | | +| pus? | Vibratory pump, 52 W | | | | +``` + +Every price gone, every quantity gone, 4 of 5 part numbers destroyed +(`MG-311`โ†’`oa`, `PMP-52`โ†’`pus?`), the `FM-9` row dropped entirely, header row +garbage. Prose paragraphs outside the table came through fine in both. + +### Defect #2 โ€” docling's markdown path loses the table + +Running the same scanned PDF through `VlmPipeline` + `API_OLLAMA`, the +`DoclingDocument` has **`tables: 0`** and the pipe table is flattened into a +single paragraph: + +``` +Part No. Description Torque (Nm) Qty Price (EUR) MG-311 Group head gasket, 8.5 mm 12.5 2 4.20 TS-71 โ€ฆ +``` + +The values are all still there and in reading order โ€” far better than tesseract's +output โ€” but the structure the model produced is thrown away by the time it +reaches `export_to_markdown()`, which is exactly what +`_perform_conversion()` in our converter consumes (it also hands the +`DoclingDocument` to structure-aware chunkers). Whether the duplication is what +breaks the markdown parse is **untested**. The raw HTTP response *does* contain +a clean pipe table, so a thin path that keeps the model's markdown would not +have this problem. + +### Latency measured here (M2 Max, scale 2.0, `num_ctx=16384`) + +| Scenario | Wall time | +|---|---| +| Cold (first request, model load) | 25.1 s | +| Warm, sequential, single page | **2.2 โ€“ 5.5 s/page** (4 consecutive runs: 8.3, 4.3, 3.7, 3.7 s) | +| Warm, 3-page PDF via docling (`concurrency=4`) | **4.38 s total โ†’ 1.46 s/page** | +| Current chain (tesseract-cli) same scanned page | 6.7 s/page | + +Render-scale sweep on the scanned invoice: `scale=1.0` โ†’ 7.1 s, all values still +correct but emitted as plain rows instead of a markdown table; `scale=2.0` โ†’ +7.3 s, proper table; `scale=3.0` โ†’ 17.1 s, **no accuracy gain**. Docling's +preset default of 2.0 is the right operating point; do not raise it. + +Upstream claims 1.86 pages/s for PDFs (Table 6, +<https://arxiv.org/html/2603.10910v1>) โ€” that is on server GPUs, not comparable, +but our 1.46 pages/s wall with concurrency 4 is in the same order. + +--- + +## Q4 โ€” Cost/benefit for localGPT + +### What our OCR chain actually resolves to on this machine (surprise) + +`build_ocr_options()` walks `OcrMac โ†’ EasyOCR โ†’ RapidOCR โ†’ tesserocr โ†’ +tesseract-cli`. Run for real: + +``` +$ .venv/bin/python -c "from rag_system.ingestion.document_converter import build_ocr_options; print(build_ocr_options())" +OCR engine: TesseractCliOcrOptions +mode=FULL_PAGE lang=['fra','deu','spa','eng'] scale=3.0 force_full_page_ocr=True +``` + +`ocrmac`, `easyocr` and `tesserocr` are not installed; `rapidocr` **3.9.2 is**, +but the probe tests for the stale module name `rapidocr_onnxruntime`, so it is +skipped. **So the premise "OcrMac fallback" is not what runs here โ€” everything +falls through to Tesseract CLI with four languages at once**, which is both the +slowest and the least accurate configuration in that list. Two much cheaper +wins than GLM-OCR exist: `pip install ocrmac` (Apple Vision, native, fast), and +fixing the RapidOCR module-name probe. Those should be measured before, or +alongside, any VLM work. + +### Where GLM-OCR would and would not move the needle + +| Document class | Verdict | +|---|---| +| Digital-born PDFs with a text layer | **No change.** The probe skips OCR entirely; GLM-OCR never runs unless we force it. Forcing it would be strictly worse (slower, and it re-flows text we already have perfectly). | +| Scanned pages, prose only | Moderate win. Tesseract handles clean prose acceptably; GLM-OCR is better on degraded input but this is not where the gap is. | +| **Scanned/photographed pages with tables** | **Large, demonstrated win** โ€” 30/30 cells vs. near-total loss (above). This is the case that justifies the whole exercise. | +| Dense tables generally | Strong on paper: TableTEDS 93.96 / TableTEDS-S 96.39, best in class on OmniDocBench v1.5 (arXiv 2603.10910). | +| Formulas | FormulaCDM 93.90 โ€” but our stack has nothing downstream that uses LaTeX, so this is not a localGPT win today. | +| **Handwriting** | **Do not promise this.** Reported 86.1 vs Gemini 3 Pro's 94.5 on handwritten KIE. Weakest category. | + +### Correction to the roadmap's evidence line + +Roadmap 1.3 says "95.22 OmniDocBench vs GPT-5.2's 86.59". The 95.22 figure is +**OmniDocBench v1.6_full**, where GLM-OCR is **third**, behind PaddleOCR-VL-1.6 +(96.34) and MinerU2.5-Pro (95.75). The "#1, 94.62" claim is the older **v1.5** +table (arXiv 2603.10910, where PaddleOCR-VL-1.5 is 94.50 โ€” a 0.12-point gap, +i.e. inside the noise our own roadmap says not to trust). I could not verify a +"GPT-5.2 86.59" row anywhere. **The roadmap row should be corrected: GLM-OCR is +an excellent small OCR model, not the uncontested leader, and PaddleOCR-VL-1.6 +now scores higher on the current benchmark.** Docling 2.118.1 also ships presets +for `lightonocr`, `dots_ocr`, `chandra_ocr2`, `falcon_ocr` and `nanonets_ocr2` โ€” +if we build a scanned-document eval, GLM-OCR should be one column in it, not the +foregone conclusion. + +--- + +## Recommendation + +### GO-LATER โ€” prototype behind a flag, do not adopt as default yet + +**Why not NO-GO:** the Apple Silicon serving path the roadmap called "unproven" +is now proven. It costs **zero new Python dependencies** (our pinned docling +already drives it), one 2.2 GB Ollama model, 2.8 GB of GPU memory while loaded, +and it fixes a failure mode that today loses *all* numeric content in scanned +tables. Warm latency (2โ€“5 s/page, 1.5 s/page with concurrency) is acceptable for +an ingestion-time, opt-in path. + +**Why not GO:** three things must land first, and none of them is code we should +write blind: + +1. **Duplication** โ€” pages are sometimes emitted twice through the Ollama build. + Needs either a de-dup post-process (cheap, ugly) or the MLX / custom-template + serving path (correct). +2. **Table structure is lost** in docling's markdownโ†’`DoclingDocument` step + (`tables: 0`), which is most of the value. Needs either the raw-markdown path + or the `ocr_options`-style plugin shape. +3. **No eval to decide with.** Phase 0 has retrieval and groundedness metrics but + **no ingestion-quality corpus**: `eval/corpora/` contains only digital-born + PDFs with clean text layers, so today nothing in the harness would even + exercise the OCR branch, let alone score it. Adopting 1.3 without that would + violate the roadmap's own decision gate. + +**Sequencing:** do the cheap OCR fixes first (`ocrmac`, RapidOCR probe name), +build a small scanned/tabular corpus with known ground truth, then A/B +GLM-OCR against the fixed baseline โ€” and against `lightonocr` / `dots_ocr`, +which are already one line away in the same preset registry. + +### Integration path when it goes ahead (config sketch, no code) + +Route **per page**, not per document โ€” today's probe is "any page has text โ†’ +no OCR for the whole document", so a scanned insert inside a digital PDF gets +nothing at all. That is a bug worth fixing independently of GLM-OCR. + +``` +# opt-in; default off โ€” the current chain stays the fallback +PARSER_VLM = off | glm-ocr # default: off +PARSER_VLM_ENDPOINT = http://localhost:11434 +PARSER_VLM_MODEL = glm-ocr +PARSER_VLM_SCALE = 2.0 # do not raise; 3.0 costs 2.3x for no gain +PARSER_VLM_CONCURRENCY = 4 +PARSER_VLM_TIMEOUT_S = 90 +PARSER_VLM_MIN_CHARS = 32 # per-page text-layer threshold +``` + +Behaviour: per-page probe โ†’ pages under `PARSER_VLM_MIN_CHARS` go to the VLM; +everything else keeps the existing text-layer path. On timeout, HTTP error, or +empty output, fall back to the current OCR chain for that page โ€” never fail the +ingest. De-duplicate identical output halves before chunking. Docling wiring is +`VlmConvertOptions.from_preset("glm_ocr", engine_options=ApiVlmEngineOptions( +engine_type=VlmEngineType.API_OLLAMA))` plus **`enable_remote_services=True`**. + +### Open risks + +- **"Remote services" flag.** Turning on a docling option literally named + `enable_remote_services` in a product called *localGPT* needs a comment in the + code and a line in the docs saying it points at `localhost:11434`. If a user + ever overrides `PARSER_VLM_ENDPOINT`, documents leave the box. +- **Ollama build ignores prompts.** We are locked to one recognition mode, and a + future Modelfile change upstream could silently alter output. Pin the tag. +- **Second runtime dependency at ingest time.** Ingestion would fail-soft but get + much slower if Ollama is down or busy serving the chat model โ€” GLM-OCR + competes for the same GPU as the generation model. +- **Ollama version floor.** The model requires Ollama โ‰ฅ 0.15.5; we would need to + state that (this host runs 0.32.6). +- **Docker.** `docker-compose.local-ollama.yml` exists, but nothing here was + tested in a container, and the 2.2 GB model would need to be present in + whichever Ollama the container talks to. +- **Benchmark drift.** OmniDocBench moved from v1.5 to v1.6_full between the + roadmap being written and this spike, and GLM-OCR moved from 1st to 3rd. Do + not re-cite leaderboard positions in our docs; cite our own eval or nothing. +- **Table loss may be a docling bug we cannot fix from config.** If the + markdownโ†’document parse is the blocker, the integration shape has to change + (raw markdown, or the `ocr_options` plugin), which is more work than the + "config sketch" above implies. + +--- + +## Reproducing this + +Scratch scripts (not in the repo): +`/private/tmp/claude-501/-Users-prompt-videos-localgpt-08082026/4d62420b-7ab2-4be1-90f2-708d7bae9146/scratchpad/` +โ€” `glmocr_smoke.py` (OpenAI-compatible endpoint), `glmocr_native.py` +(`/api/chat`, exposes `num_ctx`), `make_scan_fixture.py` (builds the scanned +invoice fixture). The pulled model `glm-ocr:latest` was kept, since it works. + +### Sources + +- <https://ollama.com/library/glm-ocr> +- <https://huggingface.co/zai-org/GLM-OCR> +- <https://github.com/zai-org/GLM-OCR> ยท <https://github.com/zai-org/GLM-OCR/blob/main/examples/ollama-deploy/README.md> +- <https://arxiv.org/html/2603.10910v1> (GLM-OCR technical report) ยท <https://arxiv.org/html/2603.10910v2> +- <https://huggingface.co/mlx-community/GLM-OCR-bf16> +- <https://github.com/ggml-org/llama.cpp/discussions/19721> ยท <https://huggingface.co/blog/ggml-org/using-ocr-models-with-llama-cpp> +- <https://docs.vllm.ai/en/stable/getting_started/installation/cpu/> ยท <https://github.com/vllm-project/vllm-metal> +- <https://github.com/DCC-BS/docling-glm-ocr> (unverified alternative) +- <https://docling-project.github.io/docling/reference/pipeline_options/> diff --git a/eval/decisions/judge-sonnet-validation-2026-08-13.json b/eval/decisions/judge-sonnet-validation-2026-08-13.json new file mode 100644 index 00000000..7e1fa000 --- /dev/null +++ b/eval/decisions/judge-sonnet-validation-2026-08-13.json @@ -0,0 +1,358 @@ +{ + "timestamp": "2026-08-13T03:36:34+00:00", + "judge": "claude sonnet subagents x3 (Agent tool), prompt v1, suffix-stripped", + "validation": { + "tp": 10, + "fn": 0, + "tn": 10, + "fp": 0 + }, + "hard": { + "correct": 18, + "n": 18 + }, + "per_case": { + "val_g01_brew_pressure": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_g02_descaling": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_g03_gasket": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_g04_e42_procedure": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_g05_warranty_length": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_g06_annual_leave": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_g07_sick_pay": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_g08_sabbatical": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_g09_public_holiday": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_g10_contractors": { + "votes": [ + true, + true, + true + ], + "set": "validation", + "label": true + }, + "val_u01_brew_pressure_wrong": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "val_u02_descale_interval_wrong": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "val_u03_boilers_swapped": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "val_u04_e42_extra_step": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "val_u05_sensor_part_transposed": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "val_u06_warranty_length_wrong": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "val_u07_annual_leave_wrong": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "val_u08_bereavement_swapped": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "val_u09_parental_transfer_added": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "val_u10_sabbatical_notice_wrong": { + "votes": [ + false, + false, + false + ], + "set": "validation", + "label": false + }, + "hard_cell1_off_acq_q04": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_off_acq_q07": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_off_acq_q09": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_off_acq_q12": { + "votes": [ + false, + false, + false + ], + "set": "hard", + "label": false + }, + "hard_cell1_off_docs_d03": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_off_docs_d05": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_off_docs_d13": { + "votes": [ + false, + false, + false + ], + "set": "hard", + "label": false + }, + "hard_cell1_on_acq_q04": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_on_acq_q07": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_on_acq_q09": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_on_acq_q12": { + "votes": [ + false, + false, + false + ], + "set": "hard", + "label": false + }, + "hard_cell1_on_docs_d03": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_on_docs_d05": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell1_on_docs_d13": { + "votes": [ + false, + false, + false + ], + "set": "hard", + "label": false + }, + "hard_cell2_off_acq_q04": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell2_off_acq_q12": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell2_on_acq_q04": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + }, + "hard_cell2_on_acq_q12": { + "votes": [ + true, + true, + true + ], + "set": "hard", + "label": true + } + } +} \ No newline at end of file diff --git a/eval/decisions/multiturn-decomposer-2026-08-16.md b/eval/decisions/multiturn-decomposer-2026-08-16.md new file mode 100644 index 00000000..0a868348 --- /dev/null +++ b/eval/decisions/multiturn-decomposer-2026-08-16.md @@ -0,0 +1,72 @@ +# Multi-turn decomposer: answer-visibility fix (arms m0โ€“m1d, L) โ€” 2026-08-16 + +**Verdict: ADOPTED as a two-variant prompt.** The decomposer now sees the last +assistant answer โ€” but only when conversation history exists. The single-turn +prompt is frozen byte-exact, because arm L measured that even benign-looking +prompt additions cost 2 Sonnet-confirmed rfc rows. + +## The gap + +`QueryDecomposer` received only the last 5 *user* queries. A follow-up whose +antecedent was introduced by the *assistant* ("who is the largest supplier?" โ†’ +answer names Acme โ†’ "what is their lead time?") had no antecedent visible. + +## Instrument: eval/goldset/multiturn.jsonl + +12 hand-authored conversations over the 5 eval corpora (8 "answer-entity" +class isolating this gap; 4 pronoun-to-user-turn controls), every expected +answer verified against chunk text. Runner +(scratchpad `multiturn/run_e2e_multiturn.py`) executes turns sequentially +through `Agent.run` with a real `session_id` โ€” turn 2 sees whatever the system +actually answered โ€” and grades the final turn. Companion harness +`multiturn/decomp_stability.py` dumps temp-0 decompositions of all 120 +single-turn gold queries; it is **perfectly deterministic run-to-run** +(120/120 byte-identical on a repeat run), so any dump diff is a real effect +of a prompt change. + +## Measurement arc + +| arm | change | multi-turn E2E | decomposition-level | +|---|---|---|---| +| m0 | baseline | 12/12 | **wrong underneath**: mt_07 resolved "their" โ†’ StartupXYZ (wrong entity); mt_09 turn 3 echoed the two previous queries verbatim. Tiny corpora + pooled synthesis-vs-original-query masked both. | +| m1 | answers interleaved into chat_history | 11/12 | mt_07 fixed (MegaCorp) but the 4b decomposer **anchored on a previous turn and substituted its query** for the current one (mt_09) | +| m1b | queries-only history + separate `last_assistant_answer` field (300-char cap) | 11/12 | mt_07 fixed; mt_09 ellipsis still collapsed to the previous question โ€” pre-existing weakness (m0 decomposed it wrong too, with luckier coverage) | +| m1c | + output rule 1b ("resolved_query must ask the SAME fact; never substitute an earlier question") + one ellipsis example (permit domain, deliberately unlike any gold row) | **12/12** | both critical rows resolve correctly | + +## Arm L: the single-turn cost that forced the two-variant design + +m1c's prompt additions reworded 25/120 single-turn decompositions (smart-quote +schema fix alone: 5/120). Full 5-bench gate (arm L vs arm K, 120 rows): + +- authored (deterministic 4b): 74โ†’75, **zero genuine down-flips** on 96 rows +- rfc: 2 real regressions, Sonnet-panel-confirmed โ€” q18 (3/3 both directions: + pre-decomposition kept two anchors "Initial-keys salt / QUIC-TLS ยง5.2" + + "QUICv2 salt"; m1c collapsed both sub-queries onto QUICv2 and lost ยง5.2), + q21 (2/3: dropped "HTTP/3" from the sub-query). q16 passes both arms (3/3). + +Iterating the prompt against those rows would be tuning to the bench. Instead: + +## Resolution: two variants, selected on `bool(chat_history)` + +- `_decompose_single_turn` โ€” the committed pre-change prompt, **byte-exact** + (including its curly-quoted schema; cosmetic fixes measurably shift temp-0 + decompositions, so the smart-quote fix lives only in the multi-turn variant). +- `_decompose_multi_turn` โ€” the m1c form. + +Verification: single-turn dump **byte-identical to the pre baseline on all +120 queries** (the arm-L rfc regression is eliminated by construction); +multi-turn confirmation arm m1d = 12/12, mt_07 โ†’ MegaCorp, mt_09 โ†’ "When was +the early termination of the waiting period granted for the HSR filing?". + +## Rules going forward + +- The single-turn prompt is a frozen measured artifact. Any byte change โ€” + cosmetic included โ€” re-triggers the full 5-bench gate. The stability dump + (`decomp_stability.py`, byte-compare vs `decomp_dump_pre.jsonl`) is the + cheap first check. +- The multi-turn prompt is gated on `multiturn.jsonl` (grow this set as + conversational failure shapes appear) + the single-turn byte-identity check. +- Artifacts: `mt_answers_{m0,m1,m1b,m1c,m1d}.jsonl`, + `decomp_dump_{pre,postq,postq2,postm1,postm1c,postm1d}.jsonl`, + `rfc_e2e_answers_l.jsonl`, `authored_e2e_answers_decomp.jsonl`, + `judged4bdet_*.jsonl`, `votes_armL_{1,2,3}.jsonl` in the session scratchpad. diff --git a/eval/decisions/multivector-retrieval-2026-08-19.md b/eval/decisions/multivector-retrieval-2026-08-19.md new file mode 100644 index 00000000..9d6b25f0 --- /dev/null +++ b/eval/decisions/multivector-retrieval-2026-08-19.md @@ -0,0 +1,98 @@ +# Multi-vector (late-interaction) first-stage retrieval โ€” NOT adopted + +**Date:** 2026-08-19 ยท **Status:** measured, not adopted ยท **Code:** env-gated hook kept (`MV_RETRIEVAL_ENDPOINT`) + +## Question + +The earlier experiment (experiments-resolveonly-maxsim-2026-08-19.md) tested MaxSim as a +*reranker* and lost. This one tests the other use from the HF multi-vector-encoder blog: +replacing the **dense leg of hybrid retrieval** with token-level MaxSim search โ€” does +multi-vector *candidate generation* beat our single-vector dense leg? + +## Model selection (from the blog's NanoBEIR NDCG@10 table) + +- `lightonai/LateOn-regularized` โ€” 149M, 128d, 0.6897 (top text-retrieval score) โ†’ arm mvA +- `LiquidAI/LFM2.5-ColBERT-350M` โ€” 353M, 128d, 0.6864 (near-tied, different family) โ†’ arm mvB + +Not run: GTE-ModernColBERT (dominated by LateOn, same lab), mxbai-edge-colbert-32m / +answerai-colbert-small (speed plays; irrelevant at our corpus sizes), mLateOn (multilingual). + +## Setup + +- Sidecar (`scratchpad/mvretrieval/server.py`) in the isolated ST6 venv (ST 6.0.0 needs + transformers 5.x / torch โ‰ฅ2.5; repo stays pinned to transformers 4.51.0 / torch 2.4.1 MPS). + Pre-encodes each bench table's `text` column (the same enriched text the dense vectors + were built from), serves brute-force MaxSim top-k over 5โ€“683 chunks per corpus. +- Repo hook: `MultiVectorRetriever._mv_sidecar_search`, env-gated on `MV_RETRIEVAL_ENDPOINT`. + Replaces only the dense leg; FTS leg, RRF fusion, Qwen3-Reranker-4B, generation all + unchanged. Raises (never silently falls back) when the env is set, so an arm cannot + quietly revert to the control. Zero extra calls when the env is unset. +- Verified active: a trap embedder in the dense path was never invoked; top hit for the + RFC 2119 smoke query was the correct definition chunk. +- Control: current defaults (latechunk off, verification off) = **90/120** 4b-det + (the nolc arm, re-confirmed by the resolve-only control). + +## Results (120-row single-turn suite, deterministic 4b judge + blind 3ร— Sonnet panel on every flip) + +| arm | model | 4b-det | flips vs control | panel-corrected net | +|-----|-------|--------|------------------|---------------------| +| mvA | LateOn-regularized | 85/120 | โ†‘1 โ†“6 | **โˆ’2** (+1 real / โˆ’3 real; 4b wrong on 3 downs) | +| mvB | LFM2.5-ColBERT-350M | 88/120 | โ†‘3 โ†“5 | **โˆ’1** (+2 real / โˆ’3 real; 4b wrong on 2 downs + 1 up) | + +14 of 15 panel cells unanimous (single split: mvA rfc_q06, resolved grounded 2-of-1). + +Real losses: mvA acq_q08, docs_d17, docs_d22; mvB docs_d08, docs_d17, rfc_q06. +Real gains: docs_d12 (both arms); mvB rfc_q24 โ€” a long-standing crossref-residue loss +that multi-vector retrieval genuinely fixed. + +## Verdict + +Neither model beats the single-vector dense leg. mvB's โˆ’1 sits at the noise floor +(1โ€“2 rows on n=24), mvA's โˆ’2 just below it โ€” but the direction is negative in both arms +and adoption is not free: token-level vectors (~30โ€“40x storage), a second model process +on a dependency stack the repo cannot host in-venv, and no latency win (the dense leg +was never the bottleneck; the reranker is). **Not adopted.** The env-gated hook stays +(zero-cost when unset) for future experiments. + +## Confounds / scope notes + +- Embedder sizes differ (dense harrier-oss-v1-0.6b โ‰ˆ 600M vs 149M/350M MV). The blog's + leaderboard has no larger maintained text MV encoder to remove this. +- Document-scoped crossref-hop searches and the overview prefilter still use the dense + path in the MV arms (shared with the control everywhere except the first stage). +- Possible ColBERT document-length truncation on long chunks was not instrumented. +- Metadata prefilters are unimplemented on the MV path (raises; bench uses none). + +## Lead worth keeping + +mvB fixing rfc_q24 while the control misses it shows MV retrieval surfaces genuinely +different candidates. If anything comes of this, it is a **third RRF leg** +(FTS + dense + MV) rather than a replacement โ€” untested, and only worth trying with +storage/process costs solved. + +## Addendum (same day, user-directed): the third-RRF-leg experiment โ€” also NOT adopted + +Arm `rrf3`: FTS + dense + LFM2.5-ColBERT MaxSim fused as THREE RRF legs +(`MV_RRF_LEG=1` alongside `MV_RETRIEVAL_ENDPOINT`; dense leg verified still firing +plus exactly one sidecar call per retrieval). Judged per user direction with +**Sonnet, not the 4b model** โ€” both arms bulk-judged by 5 blind Sonnet agents +(one per corpus), flips panel-verified by 3 more. + +| arm | Sonnet bulk | panel-corrected | +|-----|-------------|-----------------| +| control (current defaults) | 100/120 | โ€” | +| rrf3 | 100/120 | **net โˆ’1** (+2 real: acq_q01, acq_q11; โˆ’3 real: docs_d05, docs_d07, docs_d22; 36/36 panel votes unanimous; docs_d20 flip dissolved โ€” both arms ungrounded) | + +The third leg trades rows ~1:1 instead of adding recall: RRF dilution shifts fused +rankings everywhere at once. The rfc_q24 gain from replacement mode did NOT survive +dilution to a third leg (rfc identical 20/24 both arms). Verdict: not adopted; both +env-gated modes stay as documented experimental hooks. + +Calibration note: the same 120 control answers score 90/120 under the deterministic +4b judge and 100/120 under Sonnet โ€” the 4b bulk judge under-credits ~10 rows/120, +consistent with every panel correction to date. Ordering conclusions survive; exact +4b totals should not be quoted as absolute quality. + +Ops: the sidecar now persists document token-embeddings to disk +(`mvretrieval/emb_cache/`, keyed by model + exact corpus content), so encodings are +computed once per model+table and restarts reload from disk. diff --git a/eval/decisions/paraphrase-robustness-2026-08-20.md b/eval/decisions/paraphrase-robustness-2026-08-20.md new file mode 100644 index 00000000..30c23c10 --- /dev/null +++ b/eval/decisions/paraphrase-robustness-2026-08-20.md @@ -0,0 +1,64 @@ +# Paraphrase-robustness study: does multi-vector help once queries stop matching document wording? + +**Date:** 2026-08-20 ยท **Judge:** Sonnet subagents throughout (user-directed; no 4b numbers in this record) + +## Question + +The multi-vector experiments (multivector-retrieval-2026-08-19.md) lost or tied on the +standard gold sets, and the leading explanation was question style: our gold queries +reuse document vocabulary, which favors FTS+dense. User-directed test: rewrite every +gold question as a same-meaning paraphrase a person who never read the documents would +ask, then re-run the retrieval configurations. + +## The paraphrase set โ€” eval/goldset/paraphrases.jsonl + +120 rows ({id, original, para}), written by 5 Sonnet agents under rules: meaning exactly +preserved (the existing gold answer must remain the unique correct answer), maximum +vocabulary drift, identifiers replaced by unambiguous descriptions where possible. +Mean content-word Jaccard overlap with originals: **0.21**. A separate Sonnet verifier +checked all 120 pairs for answer-preservation: **0 flags**. The gold sets themselves are +untouched; the runner substitutes queries by id (PARA_QUERIES hook in the scratch runner). + +## Results (Sonnet bulk = 5 blind agents/arm; every flip vs control panel-verified by 3 voters) + +| retrieval config | original questions | paraphrased questions | +|---|---|---| +| dense + FTS (shipped hybrid) | **100/120** | **95/120** (โˆ’5: paraphrasing genuinely hurts; rfc/docs hardest hit) | +| MV replaces dense (LFM2.5-ColBERT) | 88 (panel net โˆ’1) | 90 โ€” panel net **โˆ’5 real** vs ctlp (1 gain / 6 losses, 21/21 votes unanimous; rfc 13/24) | +| FTS + dense + MV (3-leg RRF) | 100 (panel net โˆ’1) | 97 โ€” panel net **+2 real** vs ctlp (3 gains: docs_d13/d17/d18, 1 loss: rfc_q24; 3 of 8 panel cells split-vote) | + +## Findings + +1. **Paraphrasing costs the shipped pipeline 5 rows** (100โ†’95). The reworded set does + what it was built to do: shifts load from lexical matching to semantics. +2. **MV as a dense replacement fails even here** โ€” net โˆ’5 real, the worst cell measured. + The 0.6B dense embedder absorbs vocabulary drift better than the 350M ColBERT + (rfc_q01/q02/q08 lost on rewritten wording that dense handled). "It was only the + question style" is REFUTED for replacement mode. +3. **MV as a third leg is the first genuinely positive cell** (+2 real): with FTS + weakened by paraphrasing, the extra semantic leg recovers docs rows the rewording + broke (d13/d17 were paraphrase-losses; 3-leg wins them back). On original questions + the same config was net โˆ’1. Caveats: +2 on n=120 is at the noise floor, and unlike + the mvp verdict the deciding panels carry 3 split votes. + +## Decision + +Defaults unchanged: dense+FTS hybrid stays. The 3-leg's +2 appears only on +paraphrase-style queries, costs โˆ’1 on document-phrased ones, and carries the sidecar + +token-level-storage overhead. **Recommendation:** treat 3-leg (`MV_RRF_LEG=1`) as the +configuration of choice only for a deployment whose real users ask in their own words +rather than the documents' โ€” and validate on that workload first. + +## Fusion recommendation (recorded for follow-up) + +Rank-based RRF is where MV's candidates go to die: a leg that disagrees re-shuffles +everything, so gains pay a tax elsewhere (rfc_q24 won in replacement mode, lost in both +3-leg runs). If MV integration is ever pursued seriously, change the fusion, not the leg: +1. **Candidate-pool union**: feed the union of each leg's top-k to the Qwen reranker and + let it arbitrate โ€” no rank fighting, bounded by reranker latency (~linear in pool size). +2. **Weighted RRF** (down-weight the MV leg) as a cheaper middle ground. +3. Score-normalized fusion only if (1) is too slow โ€” score scales across legs are not + comparable without calibration. +Option 1 is the recommended next experiment because the reranker is this pipeline's +strongest component and the observed failure mode is exactly "right chunk found by one +leg, pushed out of the reranker window by fusion". diff --git a/eval/decisions/phase2-gateway.md b/eval/decisions/phase2-gateway.md new file mode 100644 index 00000000..c0d98832 --- /dev/null +++ b/eval/decisions/phase2-gateway.md @@ -0,0 +1,267 @@ +# Phase 2.3 โ€” cheapen gateway routing โ€” shipped 2026-08-09 + +Replaces the backend gateway's per-message LLM routing call with a deterministic +gate. Every number on this page was produced by a command run on this machine on +2026-08-09; the commands are at the bottom so each one can be re-run. + +**Scope:** `backend/server.py` (routing only), `backend/test_gateway_routing.py` +(new). Docs touched: `Documentation/architecture_overview.md` ยง1/ยง2.2/ยง3, +`Documentation/system_overview.md` ยง2.1 layer 1 (+ two consequential phrases), +`Documentation/api_reference.md` ยง1.3, `backend/README.md`. Nothing under +`rag_system/**` was modified โ€” the agent-side triage is untouched and remains the +system's single LLM routing layer. + +--- + +## 1. What was there + +`ChatHandler._should_use_rag(message, idx_ids)`, on the non-streaming path only +(`POST /sessions/<id>/messages`): + +1. No linked indexes โ†’ direct LLM. +2. Otherwise `_load_document_overviews(idx_ids)` read up to 40 overview + paragraphs off disk, `_route_using_overviews` pasted them into a prompt and + asked the **enrichment model** (`qwen3.5:4b`) for `USE_RAG` or `DIRECT_LLM`. + An unparseable reply defaulted to RAG. +3. If either step raised, `_simple_pattern_routing` decided by keyword and + length. + +The pattern fallback was worse than improvement_plan ยง2.3 records. It matched +its greeting list with `pattern in message_lower` โ€” **substring**, not word โ€” +so `'hi'` matched *w**hi**ch*, *t**hi**s* and *mac**hi**ne*, and `'ok'` matches +*b**ook***. Re-running the deleted function verbatim on eight Atlas-7 questions +(script in ยง5) routes **7 of 8 to direct LLM**: + +``` +DIRECT Which test procedure applies after replacing the pump? <- matched ['hi', 'test'] +DIRECT What is the test point voltage on the control board? <- matched ['test'] +DIRECT Where is the serial number engraved? <- matched [] +DIRECT What does this manual say about the steam boiler? <- matched ['hi'] +DIRECT How long is the Atlas-7 parts warranty? <- matched [] +DIRECT Which sensor part should be replaced when error code E11 appears? <- matched ['hi'] +RAG What pressure does the brew boiler operate at during extraction? <- matched [] +DIRECT Who manufactures the machine? <- matched ['hi'] +``` + +(`Where is the serial number engraved?` and `How long is the Atlas-7 parts +warranty?` matched no greeting at all โ€” they went direct for the *other* +reason: no `rag_indicator` keyword and under the 40-character question rule.) +This is why the function is deleted rather than patched. + +## 2. What ships now + +Two module-level functions in `backend/server.py` (module-level so they are +unit-testable without an HTTP request): + +```python +should_use_rag(message, idx_ids, force_rag=False) -> bool +is_smalltalk_or_meta(message) -> bool +``` + +The cascade is retrieval-first โ€” *escalate, don't pre-decide*: + +| Condition | Route | +|---|---| +| `force_rag` | RAG (unchanged) | +| No linked indexes | direct LLM (unchanged) | +| Whole message matches the smalltalk allowlist (โ‰ค 6 words) | direct LLM | +| Whole message matches the assistant-meta allowlist | direct LLM | +| Everything else | **RAG** | + +Both allowlists are anchored, whole-message regexes: + +* **Smalltalk** โ€” the message must consist *only* of allowlisted phrases + (greetings, thanks, farewells, acknowledgements) plus inert filler + (`there`, `again`, `so much`, โ€ฆ), **and** contain at least one *core* + phrase, **and** be at most `SMALLTALK_MAX_WORDS = 6` words. `"hello"`, + `"thanks!"`, `"got it, thanks"` match. `"Hello, what is the brew boiler + pressure?"` does not. +* **Assistant-meta** โ€” a tight list of self-referential questions + (`who are you`, `what model are you`, `which model do you use`, + `are you an AI`, `who built you`, `tell me about yourself`), anchored so + `"Who are the authors of the service manual?"` and `"What model number is + the pressure sensor?"` fall through to RAG. + +Deleted entirely: `_should_use_rag`, `_route_using_overviews`, +`_load_document_overviews` (its only caller was the router) and +`_simple_pattern_routing` โ€” 212 lines. `used_rag` is still returned on +`POST /sessions/<id>/messages`, and `force_rag` behaviour is unchanged +(it also still goes into the payload forwarded to the RAG API). + +**Why over-sending to RAG is safe.** `Agent._triage_query_async` in +`rag_system/agent/loop.py` runs on every request the gateway forwards and can +still return `direct_answer` without retrieving. A false "use RAG" therefore +costs one triage call on a model that would have been called anyway; a false +"answer directly" costs an unanswerable question. The gate is biased +accordingly. This is why the gateway can afford a gate this crude โ€” it is not +the decision-maker, it is a smalltalk filter in front of the decision-maker. + +**Evidence** (`Documentation/research/`): pre-retrieval LLM routing is the +weakest measured pattern of 2026 โ€” four ML approaches failed because "the need +for augmentation cannot be determined from the query alone"; fixed-hybrid beat +rule-based adaptive routing; cheap discriminative gates match LLM routers at +~zero cost. Roadmap item 2.3. + +## 3. Measurements + +### 3.1 Latency of the routing decision โ€” 20 sample messages + +Same 20 messages, same machine, same session overviews (the Atlas-7 service +manual index), warm-up call excluded, Ollama warm. + +| | old (`_should_use_rag`, enrichment-model call) | new (`should_use_rag`) | +|---|---|---| +| total, 20 messages | 15012.258 ms | 0.040 ms | +| mean | **750.613 ms** | **0.002 ms** | +| median | 755.507 ms | 0.002 ms | +| min / max | 729.089 / 760.750 ms | 0.001 / 0.006 ms | + +**Saving: ~750.6 ms per non-streaming message** (mean 750.613 โ†’ 0.002 ms), plus +one fewer `qwen3.5:4b` generation and one fewer overview file read per message. +The old path's cost was flat across message types โ€” even `"hello"` paid the +full ~750 ms โ€” because the LLM call happened before any decision. + +Caveats, stated plainly: this is one machine, one warm Ollama, one small +overview set (1 overview paragraph). A larger corpus makes the *old* number +worse (up to 40 overviews in the prompt), never better. The new number is +independent of corpus size. + +### 3.2 Decision agreement + +On those same 20 messages the new gate reproduced the old LLM router's decision +**20/20** (13 RAG, 7 direct): all 13 document questions RAG, and `hello`, +`hi there`, `thanks!`, `thank you so much`, `bye`, `who are you?`, +`what model are you` direct. Not a benchmark โ€” 20 messages on one corpus โ€” but +it is the same behaviour at 1/375,000th the cost. + +### 3.3 Unit tests โ€” `backend/test_gateway_routing.py` + +155/155 checks pass (exit 0). Sections: planted-fact questions route RAG; +smalltalk and assistant-meta route direct; **messages containing "test" / +"check" route RAG** (the old defect); `force_rag` honoured in every +combination; sessions with no indexes always route direct; the gate's source +contains no LLM/network call and the four deleted methods are gone from +`ChatHandler`. + +### 3.4 End-to-end smoke โ€” `eval/smoke_e2e.py` + +**25/25 assertions passed**, exit 0, wall clock 296.8 s (`.venv/bin/python eval/smoke_e2e.py`, +2026-08-09, after the change). All four planted facts (`9.2`, `TS-71`, `36`, +`drip tray`) came back with non-empty `source_documents` and a +`[Confidence: N%]` tag, and the streamed-turn save/round-trip assertions held. + +Stated honestly: **smoke sends `force_rag: True`** (`eval/smoke_e2e.py:188`), so +it exercises the gate's `force_rag` branch and the RAG forward path, not the +discriminative branch. ยง3.5 covers that gap. + +### 3.5 HTTP-level gate verification (the branch smoke does not reach) + +Backend started against a throwaway SQLite file (`DB_PATH=โ€ฆ/gate_db.sqlite`) +with `RAG_API_URL` deliberately pointed at a dead port, so a forwarded request +is visible as an error from the RAG API rather than an answer. A session was +linked to an index row; `used_rag` in the response is the gate's decision: + +``` +-- session WITH index linked -- + used_rag=True 'What pressure does the brew boiler operate at during ex' -> Error from RAG API (501)โ€ฆ [forwarded] + used_rag=True 'Which test procedure applies after replacing the pump?' -> Error from RAG API (501)โ€ฆ [forwarded] + used_rag=False 'hello' -> Hello! How can I help you today? โ€ฆ + used_rag=False 'thanks!' -> You're very welcome! โ€ฆ + used_rag=False 'who are you?' -> I'm **Qwen3.5**, the latest large language model โ€ฆ + +-- force_rag on smalltalk -- + used_rag=True 'hello' + force_rag -> Error from RAG API (501)โ€ฆ [forwarded] + +-- session with NO index -- + used_rag=False 'What pressure does the brew boiler operate at during ex' -> (direct answer) + used_rag=False 'hello' -> (direct answer) +``` + +The "test" question โ€” the old defect โ€” is forwarded to RAG over real HTTP, and +`used_rag` is still present and correct in every response body. + +### 3.6 Worst-case regex cost + +The smalltalk regex is an alternation under a `*` quantifier, so it was checked +for catastrophic backtracking: `"hi " ร— 200`, `"thanks so much and " ร— 50`, +`"who are you " ร— 40`, a 5 KB tail after a meta prefix and a 20 KB blob all +resolve in 0.004โ€“0.011 ms. The โ‰ค 6-word cap runs before the smalltalk regex, and +the meta regex is anchored at both ends, so neither can be driven into a blowup. + +--- + +## 4. Proposed `improvement_plan.md` rows + +*(This agent does not edit `improvement_plan.md` or `research_roadmap.md`.)* + +**Add to ยง0 Landed:** + +| Area | Change | Verify at | +|------|--------|-----------| +| Routing | **Roadmap 2.3 โ€” gateway routing is a deterministic gate.** The per-message enrichment-model router and the `_simple_pattern_routing` keyword/length fallback are deleted; `should_use_rag()` routes on `force_rag` โ†’ linked indexes โ†’ a whole-message smalltalk/assistant-meta allowlist โ†’ RAG. ~750 ms/message saved; agent triage is now the only LLM routing layer | `backend/server.py::should_use_rag`, `backend/test_gateway_routing.py` (155/155), `eval/decisions/phase2-gateway.md` | + +**Remove from ยง2 Routing / triage (Open):** + +| ID | Item | Why it closes | +|----|------|---------------| +| 2.3 | Retire the keyword fallback | `_simple_pattern_routing` no longer exists; the "test" misroute is covered by a regression test | + +**Amend ยง2 item 2.1** ("Embed and cache document overviews โ€” *both routers* make +an LLM call per query"): only one router does now. The item still stands for the +agent-side router, but its rationale should say so. + +**Observation, not a claim of ownership โ€” ยง2 item 2.4** ("Make `force_rag` mean +one thing: on the gateway it selects the RAG route but is not forwarded"). In +the current tree it *is* forwarded: `handle_session_chat` sets +`options["force_rag"] = True` and `_handle_rag_query` does `payload.update(options)`. +That row looks stale against the code today; whoever owns it should re-verify. + +--- + +## 5. Commands + +```bash +# unit tests (no HTTP, no LLM) +.venv/bin/python backend/test_gateway_routing.py + +# end-to-end smoke (starts both services against throwaway stores) +.venv/bin/python eval/smoke_e2e.py + +# HTTP-level gate check (ยง3.5): backend only, throwaway DB, RAG API pointed at a +# dead port so a forwarded request is unmistakable +DB_PATH=/tmp/gate_db.sqlite RAG_API_URL=http://localhost:8899 \ + .venv/bin/python backend/server.py & +# then: create a session, POST /indexes, link them, and POST messages with and +# without force_rag, reading `used_rag` out of each response. + +# latency: the old path was timed before the refactor by calling +# ChatHandler._should_use_rag on 20 messages with the Atlas-7 overviews linked; +# the new path times server.should_use_rag on the identical list. +# Harness: eval/decisions/phase2-gateway-bench.py (see below) +``` + +The benchmark harness is 60 lines: it imports `backend/server.py`, times +`should_use_rag(message, idx_ids)` over the 20 messages listed in ยง3.1 with +`perf_counter`, and reports total/mean/median/min/max. To reproduce the "old" +column, check out the pre-change `backend/server.py`, instantiate +`ChatHandler.__new__(ChatHandler)` with an `OllamaClient`, and call +`_should_use_rag(message, ["<index_id>"])` from the repo root so the relative +overview paths resolve. + +--- + +## 6. Stale elsewhere โ€” not owned by this change + +`Documentation/triage_system.md` still documents the deleted gateway router in +detail (`_should_use_rag`, `_load_document_overviews`, `_route_using_overviews`, +`_simple_pattern_routing`, with line numbers) at lines 4, 36, 39โ€“40, 42, 82 and +101. That file is owned by another agent in this phase and was deliberately not +touched here. Every gateway-routing statement in it is now false and needs the +ยง2 cascade above. + + +--- + +**Gate resolution (2026-08-09):** ยง6's flag about `Documentation/triage_system.md` is +resolved โ€” the file was rewritten at the validation gate; every gateway-routing +statement now describes `should_use_rag()`. diff --git a/eval/decisions/phase2-pipeline.md b/eval/decisions/phase2-pipeline.md new file mode 100644 index 00000000..b02a21cc --- /dev/null +++ b/eval/decisions/phase2-pipeline.md @@ -0,0 +1,394 @@ +# Phase 2 pipeline-shape items โ€” 2.5, 2.1, 2.2, 2.4 + +_Investigated and shipped 2026-08-09._ + +The four items this page covers, from +[`Documentation/research_roadmap.md`](../../Documentation/research_roadmap.md) ยงPhase 2: + +| # | Item | Outcome | +|---|------|---------| +| 2.5 | Delete the graph module | **Removed.** Code, config keys, two dependencies, and every doc section | +| 2.1 | Evidence-sufficiency retry | **Shipped ON** in `default`, off in `fast`. Fires on 9.7โ€“11.1% of `mixed`, +0.008 to +0.017 nDCG@10 and +0.014 recall@10, zero per-query regressions across four runs | +| 2.2 | Decomposition at rerank | **Shape change shipped; sub-query scoring measured NEGATIVE (โˆ’0.046 `max`, โˆ’0.012 `mean`) on the 6 affected queries and is enabled by no shipped profile.** First stage no longer fans out over sub-queries except behind `compose_from_sub_answers` | +| 2.4 | Verifier model seam | **Seam shipped, default unchanged.** ThinknCheck has no public weights; two suitable substitutes were found, wired and smoke-tested | + +Every number below was produced by running `eval/run_eval.py` on this tree. Where +a measurement is missing or was not affordable, this page says so rather than +estimating. + +--- + +## 0. The baseline, and an honest note about corpus drift + +The `docs` and `mixed` corpora index live `Documentation/*.md`, so **this work's +own documentation edits changed the corpus underneath the metric.** Every run +below therefore names its chunk count, and only same-chunk-count runs are +compared to each other. + +| Run | chunks (`mixed`) | mixed nDCG@10 | mixed r@10 | docs nDCG@10 | docs r@10 | +|---|---|---|---|---|---| +| Pre-change tree (snapshotted first, `phase2_before.json`) | 317 | 0.913 | 0.958 | 0.753 | 0.875 | +| Settled tree, retry **off** (`phase2_final_retry_off.json`) | 331 | 0.888 | 0.958 | 0.680 | 0.875 | +| Settled tree, shipped defaults, run 1 | 331 | 0.896 | 0.958 | 0.735 | 0.917 | +| **Settled tree, shipped defaults, run 2** (`phase2_after.json`) | 331 | **0.901** | **0.972** | **0.735** | **0.917** | + +Two shipped-defaults runs are listed because the retry makes one LLM call when it +fires, so the arm is not bit-reproducible. The spread (0.896โ€“0.901 on `mixed`) is +the size of that nondeterminism; the firing *set* is identical between them. + +The pre-change 0.913 reproduces `DECISIONS.md` ยง4 exactly, so the snapshot is +sound. The 0.913 โ†’ 0.888 gap is **entirely corpus-side**, and that is checkable +rather than asserted: with the retry off, the first-stage code path is unchanged, +and of the 72 `mixed` queries **only 11 moved at all โ€” every one of them +docs-anchored, zero PDF-anchored**. The graph-removal rewrites plus the new +retry/decomposition/verifier sections added 14 chunks of fresh distractor prose to +a 317-chunk corpus. + +### One gold row was orphaned by item 2.5 + +`docs_d10` asks *"How many model calls does knowledge-graph extraction spend on +each chunk?"* and anchors on the string `"makes two LLM calls per chunk"`, which +lived **only** in the `indexing_pipeline.md` knowledge-graph section that item 2.5 +deleted. The row is now structurally unanswerable. The harness reports it as a +coverage failure on both corpora rather than hiding it, which is Gate 2 working as +designed โ€” but it also scores 0 by construction and drags every future run down by +~0.014 on `mixed` and ~0.042 on `docs`. + +**It is left in place, not quietly edited** โ€” `eval/goldset/` is not this work's to +change, and silently repairing a gold row to make one's own change look better is +exactly the failure mode the honesty rule exists to prevent. Excluding it: + +| Run (mixed, n=71, docs_d10 excluded) | nDCG@10 | recall@10 | +|---|---|---| +| Pre-change tree | 0.9114 | 0.9577 | +| Settled tree, retry off | 0.9006 | 0.9718 | +| Settled tree, shipped defaults, run 1 | 0.9086 | 0.9718 | +| **Settled tree, shipped defaults, run 2** | **0.9141** | **0.9859** | + +| Run (docs, n=23, docs_d10 excluded) | nDCG@10 | recall@10 | +|---|---|---| +| Pre-change tree | 0.7427 | 0.8696 | +| Settled tree, retry off | 0.7092 | 0.9130 | +| **Settled tree, shipped defaults** | **0.7668** | **0.9565** | + +So the shipped stack scores **0.9086โ€“0.9141 on `mixed` against a pre-change 0.9114** +โ€” i.e. it straddles the baseline, well inside a noise floor of one query โ‰ˆ 0.014, +while carrying 14 extra distractor chunks. Recall@10 is up in both runs +(0.9577 โ†’ 0.9718/0.9859), and `docs` **beats** the pre-change baseline outright +(+0.024 nDCG@10, +0.087 recall@10). **Nothing regressed.** + +--- + +## 1. Item 2.5 โ€” the graph module is gone + +Deleted: `rag_system/indexing/graph_extractor.py`; `GraphRetriever` +(`retrieval/retrievers.py`); `GraphQueryTranslator` (`retrieval/query_transformer.py`); +the extraction block and `networkx` import in `pipelines/indexing_pipeline.py`; the +`graph_strategy` constructor branch, `_run_graph_query()` and the `graph_query` +routing branch in `agent/loop.py`; and the `retrieval.graph` keys in `main.py`. + +`networkx`, `fuzzywuzzy` and `python-Levenshtein` had no other consumer once +`GraphRetriever` went (verified by grep across `*.py`) and were removed from +`requirements.txt`, `requirements-docker.txt` and `rag_system/requirements.txt`. + +**Triage is now two-way.** `Agent._normalize_triage()` runs on every verdict from +both routers and maps anything that is not an explicit `direct_answer` to +`rag_query`, so a small utility model that still emits the retired `graph_query` +label lands on the RAG path instead of a `hasattr` check that no longer exists. + +**Evidence** (`Documentation/research/academic-evidence-2026.md` ยง6): GraphRAG +*loses* on single-hop retrieval (64.78 vs 63.01 F1 on NQ; 60.92% vs 60.14% on +GraphRAG-Bench); its multi-hop gains span +3 to +27 points depending entirely on +how well the vector baseline is tuned; and it costs **41โ€“57ร— at indexing** +(135s โ†’ 5,560โ€“7,702s) and up to **~377ร— in query tokens** (879 โ†’ 331,375 prompt +tokens/query for MS-GraphRAG global). It was also unreachable โ€” no shipped profile +ever set `graph_strategy`. + +Docs updated in the same change: `retrieval_pipeline.md`, `indexing_pipeline.md`, +`system_overview.md`, `architecture_overview.md`, `triage_system.md`, +`prompt_inventory.md`, `verifier.md`, `docker_usage.md`, `rag_system/README.md`, +`rag_system/DOCUMENTATION.md`, `README.md`, `DOCKER_README.md`. + +`improvement_plan.md` ยง9's graph bullet resolves to **removed** (proposed row in ยง5). + +--- + +## 2. Item 2.1 โ€” evidence-sufficiency retry + +### 2.1 The signal, and the one that did not work + +The roadmap says to trigger on the top score. Measured on the gold set, **the raw +top cosine similarity is anti-correlated with success and is unusable**: + +| `mixed`, top cosine | value | +|---|---| +| The 3 first-stage misses | 0.609, 0.629, 0.642 | +| Successful queries | min 0.441, p25 0.544, **median 0.576**, max 0.753 | + +All three failures score *above* the median success. Any threshold catching all +three fires on **94% of successful queries**. Absolute similarity mostly encodes how +close a query's phrasing sits to the corpus register, not whether the answer was +found. RRF scores are worse still โ€” every query's top RRF is 0.031โ€“0.033. + +What does carry signal is **contrast** โ€” how far the best candidate stands above the +background of everything else the query pulled in: + +``` +evidence = (cos_top โˆ’ cos_background) / (1 โˆ’ cos_background) +``` + +with `cos_background` the mean cosine of candidates from rank 6 down, and +`cos = 1 โˆ’ _distance/2` on the L2-normalized v4 tables. The denominator rescales +against this query's reachable headroom, keeping the score in 0โ€“1 and comparable +across queries. + +### 2.2 Calibration + +| threshold | `mixed` fails caught | `mixed` successes fired | `mixed` fire rate | +|---|---|---|---| +| 0.10 | 1/3 | 3/69 (4.3%) | 5.6% | +| **0.12 โ† shipped** | **1/3** | **5/69 (7.2%)** | **8.3%** | +| 0.14 | 1/3 | 10/69 (14.5%) | 15.3% | +| 0.20 | 3/3 | 18/69 (26.1%) | 29.2% | + +**The brief's target โ€” fire on the genuine failures without firing on >10% of +successes โ€” is not achievable on this gold set, and that is a finding, not a tuning +failure.** Catching all three misses costs 26% false-fire on `mixed` and 67% on +`docs`. `0.12` was chosen as the largest threshold that respects the โ‰ค10% budget; +it catches one of the three real misses. + +### 2.3 Measured effect + +Retry off vs on, same tree, same corpus snapshot. Two on-runs, because the +reformulation is an LLM call: + +| Arm | mixed nDCG@10 | mixed r@10 | docs nDCG@10 | docs r@10 | mixed fired | docs fired | +|---|---|---|---|---|---|---| +| off (mid-work tree, 313 ch) | 0.8889 | 0.9583 | 0.6781 | 0.8750 | โ€” | โ€” | +| on, run 1 (same tree) | 0.9011 | 0.9722 | 0.7096 | 0.8750 | 7/72 (9.7%) | 5/24 (20.8%) | +| on, run 2 (same tree) | 0.9063 | 0.9722 | 0.7304 | 0.9167 | 7/72 (9.7%) | 5/24 (20.8%) | +| off (settled tree, 331 ch) | 0.8881 | 0.9583 | 0.6796 | 0.8750 | โ€” | โ€” | +| on, run 1 (settled tree) | 0.8960 | 0.9583 | 0.7348 | 0.9167 | 7/72 (9.7%) | 5/24 (20.8%) | +| on, run 2 (settled tree) | 0.9014 | 0.9722 | 0.7348 | 0.9167 | 8/72 (11.1%) | 6/24 (25.0%) | + +Compare only within a chunk count. Both tree snapshots give the same verdict: the +retry is positive on both corpora and on both metrics, four runs out of four. + +* **Firing rate: 9.7% on `mixed`** on the tree the threshold was calibrated + against โ€” inside the โ‰ค10% budget โ€” drifting to 11.1% on the settled tree as the + corpus grew. That drift is the honest caveat on the threshold: it is a property + of the corpus, not a constant. +* The firing *set* is deterministic for a given corpus; only the rewrites vary, + which is why the two runs on each tree differ. +* **Zero per-query regressions on `mixed`** in any run. Three queries improved. +* It repaired `docs_d16` (*"Does this project build an approximate-nearest-neighbour + indexโ€ฆ"*), a genuine recall@10 = 0 miss, by rewriting it to *"approximate nearest + neighbor (ANN) data structure implementation and vector search executionโ€ฆ"*. +* Cost: one enrichment-model call plus one extra retrieval, on ~10% of queries. + `mixed` mean latency 96 ms โ†’ 186 ms averaged over all queries. + +**Verdict: ship enabled in `default`, disabled in `fast`.** The delta is positive on +both corpora and no query got worse. It is small โ€” +0.008 to +0.017 nDCG@10 on +`mixed` โ€” and it is bought with an LLM call, so it belongs in the quality profile +and not in the speed one. + +Caveats worth carrying: `atlas7` and `hr` are 1- and 2-chunk tables, so the +background term is degenerate there and `hr` fires on 62.5% of queries. It is +harmless (the better result set is kept either way) but those rows measure nothing. +The retry is inert on legacy unnormalized tables and in `fts_only` mode, by design โ€” +no signal, no retry. + +--- + +## 3. Item 2.2 โ€” decomposition at rerank, not first stage + +### 3.1 What was happening before + +Sub-query fan-out **did** happen at the first stage. `Agent._run_async` submitted +one full `RetrievalPipeline.run()` per sub-query to a 3-worker pool, for both the +`compose_from_sub_answers` path and the aggregate path. + +### 3.2 What runs now + +| `query_decomposition` | reranker | First stage | Rerank scored against | +|---|---|---|---| +| off | either | once, full query | full query | +| on, `compose_from_sub_answers: true` (profile default) | either | **once per sub-query**, parallel | that sub-query | +| on, `compose_from_sub_answers: false` | on | once, full query | **all sub-queries, aggregated** | +| on, `compose_from_sub_answers: false` | off | once, full query | โ€” (no rerank stage; sub-queries unused) | +| on, one sub-query after decomposition | either | once, the resolved query | the resolved query | + +The `compose_from_sub_answers` path keeps first-stage fan-out **behind its existing +flag**, as the brief allows: it needs a separate *answer* per sub-question to +compose from, which one shared candidate set cannot produce. Everything else now +retrieves once on the full original query. + +### 3.3 Measurement + +`docs` (24 queries), `Qwen/Qwen3-Reranker-4B`, retry off so it does not confound +the arms. `mixed` with the reranker on was **not run**: at ~12.5 s/query it is +~15 minutes per arm ร— 3 arms, and the brief explicitly allows skipping it. + +| Arm | first-stage nDCG@10 | **post-rerank nDCG@10** | rerank ms/query | +|---|---|---|---| +| decompose off | 0.6796 | **0.8377** | 12,471 | +| decompose on, `max` | 0.6796 | 0.8415 | 16,984 | +| decompose on, `mean` | 0.6796 | **0.8499** | 16,755 | + +**The first-stage number is byte-identical across all three arms**, which is the +structural check that matters: decomposition provably never touches the first +stage any more. A fourth arm โ€” decompose on with the **reranker off** โ€” reproduced +the off arm exactly (0.680, no rerank stage), confirming the documented no-op. + +### The headline number is confounded; the honest split is worse + +Only **6 of 24** queries decompose into more than one sub-query. The other 18 +return a single (pronoun-resolved) sub-query, and those are *not* testing +decomposition at all โ€” they are testing "rerank against the decomposer's rewrite +of the query". Splitting them: + +| Subset | off | `max` | `mean` | +|---|---|---|---| +| All 24 | 0.8377 | 0.8415 | 0.8499 | +| **The 6 genuinely decomposed** | **0.8862** | **0.8406 (โˆ’0.046)** | **0.8740 (โˆ’0.012)** | +| The 18 single-sub-query (rewrite only) | 0.8215 | 0.8418 (+0.020) | 0.8418 (+0.020) | + +**On the queries decomposition actually affects, scoring against sub-queries at +rerank is negative under both aggregates.** Two queries carry it: `docs_d14` +(0.500 โ†’ 0.431 under `max`) and `docs_d17` (0.818 โ†’ 0.613 under both). The whole- +corpus "gain" comes entirely from the single-sub-query rows, where the win is +query *rewriting*, not decomposition โ€” and even there only 2 of 18 rows moved. + +### Verdict + +* **The shape change ships.** First-stage retrieval always uses the full original + query. This is the part the evidence supports, and it is a strict reduction in + work: the aggregate path used to issue N first-stage retrievals and now issues + one. +* **Sub-query scoring at rerank is not switched on anywhere by default.** The + `default` profile ships `compose_from_sub_answers: true`, which never reaches + the aggregation path. Nothing in a shipped profile enables it. +* **`mean` is the default aggregate**, because it beat `max` on every subset + measured (โˆ’0.012 vs โˆ’0.046 where it matters). Less bad, not good. +* n_effective = 6 queries on one corpus. This measurement is too small to call + the 2026 MultiConIR/SSRB finding wrong; it is big enough to say **it did not + reproduce here**, so nothing was turned on because of it. + +--- + +## 4. Item 2.4 โ€” verifier model seam + +### 4.1 Availability, checked 2026-08-09 against the HuggingFace Hub API + +| Candidate | Verdict | +|---|---| +| **ThinknCheck** (arXiv 2604.01652, UPenn; 1B, 78.1 BAcc on LLMAggreFact) | **No public weights.** The paper is real and checks out, but a Hub search for `thinkncheck` returns **zero** models and the paper links no release. **Cannot be wired.** | +| `ibm-granite/granite-guardian-3.3-8b` | Exists, Apache-2.0. **8B / ~16 GB** โ€” an order of magnitude over the "small local verifier" budget. | +| `ibm-granite/granite-guardian-hap-38m` | Exists, 38M, Apache-2.0 โ€” but it is a **hate/abuse/profanity RoBERTa classifier**. Wrong task entirely: it does not score answer-vs-evidence. | +| `MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli` | โœ… MIT, **369 MB**, no custom code. Generic NLI. | +| `lytang/MiniCheck-DeBERTa-v3-Large` | โœ… MIT, **1.74 GB**, no custom code. Purpose-built grounded claim verification โ€” the baseline ThinknCheck itself benchmarks against. | +| `vectara/hallucination_evaluation_model` (HHEM-2.1-open) | Apache-2.0, 438 MB, but ships custom modelling code โ€” gated behind `VERIFIER_TRUST_REMOTE_CODE=1`. | + +So the roadmap's two named candidates both fail โ€” one has no weights, the other is +either too big or the wrong task โ€” but **two suitable substitutes do exist**, so the +seam was wired *and* exercised rather than left as a stub. + +### 4.2 What shipped + +`VERIFIER_MODEL` env var / `verification.model` config key, plus +`verification.threshold` (default 0.5). Unset โ‡’ the LLM-prompt verifier, unchanged. +Set โ‡’ `LocalNLIVerifier` loads the model lazily on first use, splits the answer into +sentences, scores each against the retrieved evidence as premise, and takes the +**minimum** โ€” one unsupported sentence makes the answer ungrounded, matching the +binary semantics `eval/judge.py` already uses. + +A model that cannot be loaded **raises**, printing the table above, rather than +falling back to the LLM prompt. A verifier that silently is not the verifier you +configured is worse than an error. + +### 4.3 Smoke test on `judge_validation.jsonl` + +The brief asked for a 5-case smoke test. Both models passed 5/5 on a balanced +5-case sample, so all 20 hand-labelled cases were run โ€” it costs a minute once the +weights are cached and 5 cases cannot distinguish the two: + +| Verifier | agreement | TPR (grounded, n=10) | TNR (ungrounded, n=10) | notes | +|---|---|---|---|---| +| `lytang/MiniCheck-DeBERTa-v3-Large` (1.74 GB) | **19/20** | **10/10** | 9/10 | one false *positive*: `u03_boilers_swapped` scored 63% โ€” it did not notice the two boilers' pressures had been swapped | +| `MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli` (369 MB) | 18/20 | 8/10 | **10/10** | two false *negatives*: `g07_sick_pay` (22%) and `g10_contractors` (2%) โ€” both grounded answers it called unsupported | + +The two fail in opposite directions, which is the useful result: MiniCheck is the +more permissive of the pair and misses a swapped-entity error, the generic NLI model +is stricter and rejects two correct answers. Neither is a drop-in improvement on the +LLM-prompt verifier without its own validation run โ€” `eval/judge.py --validate` +reports TPR/TNR for the judge, and the same discipline should apply before any of +these becomes a default. **Nothing was made the default here.** + +One practical note: `MiniCheck-DeBERTa-v3-Large` ships only `pytorch_model.bin` (no +safetensors) and its 1.74 GB blob stalled twice on first download in this +environment before completing; the 369 MB model is the faster thing to try first. + +Reproduce: + +```bash +VERIFIER_MODEL=MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli \ + .venv/bin/python -c " +from rag_system.agent.verifier import LocalNLIVerifier; import json, os +v = LocalNLIVerifier(os.environ['VERIFIER_MODEL']) +for r in (json.loads(l) for l in open('eval/judge_validation.jsonl')): + print(r['id'], r['label_grounded'], v.verify(r['question'], chr(10).join(r['evidence']), r['answer']).is_grounded) +" +``` + +`[Confidence: N%]` remains **UX, not a calibrated measurement**, and +`Documentation/verifier.md` now says so in a callout. Changing the backend changes +where the number comes from; it does not calibrate it. + +--- + +## 5. Proposed `improvement_plan.md` Landed rows + +For the gate to graduate โ€” this page does not edit `improvement_plan.md` or +`research_roadmap.md`. + +| Area | Change | Verify at | +|------|--------|-----------| +| Retrieval | **2.5 Graph module removed** โ€” `GraphExtractor`, `GraphRetriever`, `GraphQueryTranslator`, the `graph_query` triage outcome and the `retrieval.graph` / `graph_strategy` config keys are gone; `networkx`, `fuzzywuzzy` and `python-Levenshtein` dropped from all three requirements files. Contested gains, 41โ€“57ร— indexing and up to ~377ร— query-token cost (`Documentation/research/academic-evidence-2026.md` ยง6) | `rag_system/indexing/` has no `graph_extractor.py`; `requirements.txt`; `eval/decisions/phase2-pipeline.md` ยง1 | +| Retrieval | **2.1 Evidence-sufficiency retry** โ€” one conditional second retrieval on weak evidence, on in `default` and off in `fast`, triggered by candidate-set *contrast* rather than raw top similarity (which measured anti-correlated with success). Fires on 9.7โ€“11.1% of `mixed`, +0.008โ€“0.017 nDCG@10 and +0.014 recall@10, zero per-query regressions across four runs | `rag_system/pipelines/retrieval_pipeline.py::retrieve_candidates`, `rag_system/main.py` `retrieval.retry`; numbers in `eval/decisions/phase2-pipeline.md` ยง2 | +| Retrieval | **2.2 Decomposition applied at rerank** โ€” the first stage always runs once on the full original query; sub-queries score candidates at the rerank stage, aggregated by `query_decomposition.rerank_aggregate`. First-stage fan-out survives only behind the pre-existing `compose_from_sub_answers` flag | `rag_system/pipelines/retrieval_pipeline.py::_rerank_stage`, `rag_system/agent/loop.py`; numbers in `eval/decisions/phase2-pipeline.md` ยง3 | +| Verification | **2.4 Verifier model seam** โ€” `VERIFIER_MODEL` / `verification.model` swaps the LLM-prompt verifier for a local NLI/verifier model; default unchanged. ThinknCheck has no public weights and Granite Guardian is either 8B or the wrong task, so two verified substitutes were wired and smoke-tested instead | `rag_system/agent/verifier.py::LocalNLIVerifier`, `Documentation/verifier.md`; numbers in `eval/decisions/phase2-pipeline.md` ยง4 | +| Eval | `eval/run_eval.py` now drives `RetrievalPipeline.retrieve_candidates()` instead of calling the retriever directly, so first stage, rerank and retry are all the shipped code path; `--retry`, `--decompose` and `--aggregate` added | `eval/run_eval.py`, `eval/README.md` | +| Hygiene | ยง9's Graph-RAG bullet ("finish it or delete it") resolves to **deleted** | `eval/decisions/phase2-pipeline.md` ยง1 | + +### Not resolved here + +`docs_d10` in `eval/goldset/docs.jsonl` is orphaned by item 2.5 and needs retiring or +re-anchoring by whoever owns the gold set (ยง0). + +--- + +## 6. Limits of this evidence + +* **72 English queries on one laptop.** One query is ~0.014 nDCG@10 on `mixed`. The + retry's +0.008 is *inside* that noise floor on `mixed`; the reason to ship it is + that it is positive on both corpora across three runs with zero regressions, not + that any single delta is significant. +* **The retry's calibration set is the same 72 queries it is evaluated on.** With + three first-stage misses to calibrate against, the threshold is fitted to a + handful of points. Treat 0.12 as a starting value, not a tuned constant. +* **`atlas7` and `hr` are 1- and 2-chunk tables.** Their contrast scores are + meaningless and their rows are plumbing checks, not measurements. +* **Latency numbers are from a shared GPU** on an M2 Max and are indicative only. +* **Nothing here measures answer quality.** These are retrieval metrics; the + verifier smoke test is 20 hand-labelled cases, which is a small sample. + + +--- + +**Gate resolution (2026-08-09):** the orphaned gold row `docs_d10` flagged above was +re-anchored at the validation gate (embedder-identity-guard prose, topic +`graph` โ†’ `index_safety`, recorded in the row's `verification` field). The eval +numbers in this file predate that repair. Also fixed at the gate: the OCR probe's +stale `rapidocr_onnxruntime` module name (Q4 of the GLM-OCR spike), an explicit +`lang=['english']` (+ `OCR_LANG` env) for RapidOCR replacing docling's +`['chinese']` default, and `ocrmac` installed so macOS resolves `OcrMacOptions`. diff --git a/eval/decisions/phase4-answer-quality.md b/eval/decisions/phase4-answer-quality.md new file mode 100644 index 00000000..e8cf4d3e --- /dev/null +++ b/eval/decisions/phase4-answer-quality.md @@ -0,0 +1,530 @@ +# Phase 4 items 4.1 and 4.2 โ€” end-to-end ANSWER-QUALITY A/Bs + +Date: 2026-08-09/10 +Status: **both A/Bs run end-to-end and judged. Neither produces a clean win.** +Scope of this wave: `eval/decisions/phase4-answer-quality.md` (this file) only. +No `rag_system/`, no `eval/*.py`, no `Documentation/`, no `main.py` edits. Every +script that produced a number below lives in a scratch directory, not in the +repo; every index was built in that scratch directory and +`eval/.eval_indexes/**` was never read or written. + +Proposed calls, stated up front. **The gate decides; this file only supplies +numbers.** + +| Item | Proposed call | One-line reason | +|---|---|---| +| 4.1 full-document escalation | **HOLD โ€” do not adopt on this evidence, do not discard the code** | It moved 2 of 7 fired queries from fail to pass on the realistic corpus, but that lift is confounded with prompt truncation (see ยง6), the one fire under the shipped product default was a *harm* (4/5 โ†’ 1/5), and n is 7. | +| 4.2 cross-reference hop | **REJECT at the shipped `retrieval_k`; hold the code** | At `k = 20` it fires 0 times on all 11 `requires_crossref` rows, and where it is forced to fire (`k = 5`) it lands on the wrong document 11 times out of 11 and is a judged wash (5/11 โ†’ 5/11) with two individual harms. | + +--- + +## 1. Setup โ€” everything below ran on this machine + +**Branch** `rearchitect/evidence-gated-aug-2026`. **Interpreter** `.venv/bin/python`. +**Ollama** `localhost:11434`, generation `qwen3.5:9b`, enrichment/judge +`qwen3.5:4b`. A second agent was hammering the same Ollama for the whole +session, so **no wall-clock number in this file means anything** and none is +used as evidence. + +### 1.1 Indexes (throwaway, scratch-only) + +Built with `IndexingPipeline` on the shipped `default` profile โ€” contextual +enrichment ON, document overviews ON, late chunking ON, `extract_crossrefs` ON โ€” +with `storage.lancedb_uri` and `overview_path` pointed at the scratch directory +and `chunking.chunk_size = 512` (the value `eval/run_eval.py` uses, so chunk +counts line up with the tracked eval indexes; the profile's own default of 1500 +would put most acquisition PDFs in a single chunk and make "escalate to the full +document" a no-op by construction). + +| index | files | chunks | crossrefs extracted | resolved | +|---|---|---|---|---| +| `acq` (`eval/corpora/acquisition/*.pdf`) | 10 | 13 | 68 | **34** | +| `acqdocs` (same 10 PDFs + `Documentation/*.md`) | 24 | 373 | 215 | 93 | +| `atlas7` | 1 | 1 | 0 | 0 | +| `hr` | 1 | 2 | 0 | 0 | + +`acq` resolving **34** references matches the number the gate recorded after the +`normalize_name` prefix-strip fix (`phase4-crossref-prefilter.md`, "Gate +correction (2026-08-09)"), verified by reading +`metadata โ†’ ["metadata"]["crossrefs"]` straight out of the built LanceDB table. +9 of 10 documents are linked; edges are sensible +(`due_diligence_report โ†’ regulatory_approval`, +`risk_assessment โ†’ closing_checklist`, โ€ฆ) and no chunk self-resolves. + +**`atlas7` and `hr` were excluded from both A/Bs.** At `chunk_size = 512` they +are 1 and 2 chunks. "The top-ranked chunk's whole document" *is* the chunk +already in the synthesis context, so the two arms cannot differ for any reason +worth reporting. Stating that as a corpus limitation is more honest than +reporting 48 near-identical runs. + +### 1.2 How the agent was driven + +In-process `Agent(pipeline_configs=cfg, llm_client, ollama_config)` built from a +config dict this harness controls, so `retrieval.document_escalation.enabled` +and `retrieval.crossref_hop.enabled` flip cleanly per arm. `force_rag=True` +(triage skipped โ€” it is an LLM coin-flip that has nothing to do with either +feature), `session_id=None`, and the agent's semantic cache is cleared before +every query. No HTTP server was started, so none was left running. + +Recorded per query: the answer, `source_documents` (with `via_crossref`), +`token_usage`, every `event_callback` payload (`document_escalation`, +`crossref_hop`, `retrieval_retry`, โ€ฆ), and the full pipeline stdout. + +### 1.3 Two deliberate deviations from the shipped defaults โ€” both stated, both symmetric across arms + +**(a) `think: false` on the generation model.** With the shipped defaults the +9b generation model spends its entire context window on chain-of-thought and +returns an **empty** answer. Verbatim from the first run of this wave +(`acq_q01`, escalation on, no other change): + +``` +"answer": " [Confidence: 90%] [Warning: Low confidence. Groundedness: False]" +"token_usage": {"by_stage": {"synthesis": {"prompt_tokens": 9351, + "output_tokens": 7033, "calls": 1}}} +``` + +`9351 + 7033 = 16384` exactly โ€” the whole context window, no `response` text at +all. Every arm in this file therefore monkeypatches `OllamaClient` in the +*scratch runner* to default `enable_thinking=False` (the repo already does this +for `format="json"` calls; nothing in `rag_system/` was edited). This is a +finding in its own right and is logged as backlog in ยง8. + +**(b) query decomposition OFF in the `acqdocs` arms and in every 4.2 arm.** +Kept ON (profile default) for the `acq` escalation A/B in ยง3. Turned OFF +elsewhere for one reason: with it on, the string that actually reaches the +retriever is the decomposer's rewrite, which is a fresh LLM sample every run, so +the *fire set itself* stops being reproducible (see ยง3.2 for the case where it +also broke retrieval outright). Both arms of every A/B share the setting. + +### 1.4 The judge + +`eval/judge.py`'s Phase-0.3 `GroundednessJudge`, prompt `v1`, `qwen3.5:4b` โ€” +**reused unmodified**, no new judge was written. Re-validated this session: + +``` + prompt v1 on qwen3.5:4b + confusion TP=10 FN=0 TN=10 FP=0 unparseable=0 + TPR 1.0 TNR 1.0 overall agreement 1.0 (gate: >= 0.90) +``` + +(The JSON that run wrote into `eval/results/` was deleted โ€” this wave owns one +file.) + +**Slot assignment, which matters:** `EVIDENCE` = the *system's* answer, +`ANSWER` = the *gold* answer, `QUESTION` = the gold query. So `grounded: true` +means **"the system's answer contains the gold fact"**. The opposite assignment +fails every correct-but-verbose answer, because v1 rejects any claim the +evidence does not state and the system answers run 5โ€“20ร— longer than the +one-sentence gold answers. Consequences, stated plainly: extra material in the +system answer is tolerated by construction, and an answer that states the gold +fact *and* contradicts it elsewhere would still pass. It is a gold-fact-recall +measure, not a precision measure. + +**Judge nondeterminism is much worse in this framing than on the validation +set.** The brief warned about one flip in twenty; on the first two-run pass over +the 24 `acq` rows the judge disagreed with itself on **5/24** and **7/24** rows +(`14/24` vs `9/24`, and `12/24` vs `9/24`). Every judged cell below was +therefore re-run **five times** and is reported as `k/5`, with the raw first two +runs shown per row so the two-run protocol is still visible. A row counts as a +pass at majority (`โ‰ฅ3/5`). One verdict in the whole session came back +unparseable (1 of 550 five-run verdicts, in `acq_esc_off`); it is counted as +not-a-pass. No arm produced a literally empty answer once thinking was off; the +abstention string *"I could not find that information in the provided +documents."* appears 8 times across 110 judged runs and is judged like any other +answer. + +--- + +## 2. Fire-subset discovery for 4.1 + +Two screens, because they disagree and the disagreement is informative. + +**Retrieval-only screen** (`EscalatingRetrievalPipeline.retrieve_candidates`, +no synthesis, raw gold query): reads the same post-retry evidence signal the +escalation planner reads. + +| corpus | n gold | fires at the shipped threshold (`retry.min_top_score` = 0.12) | fire rate | +|---|---|---|---| +| `acq` | 24 | 2 (`acq_q04` 0.1072, `acq_q12` 0.1072) | 8.3% | +| `acqdocs` | 48 | 9 (`acq_q04` .113, `acq_q07` .1129, `acq_q09` .0924, `acq_q12` .0566, `docs_d03` .0926, `docs_d05` .1189, `docs_d13` .080, `docs_d15` .0916, `docs_d20` .0955) | 18.8% | + +Signal was `dense_contrast` on every single query (reranking is off in the +shipped profile, so the calibrated-reranker branch never runs). + +**Agent screen** โ€” the authoritative one, because only a real +`Agent.run()` emits `document_escalation`: + +| arm | corpus | decomposition | n | queries that actually escalated | +|---|---|---|---|---| +| `acq_esc_on` | `acq` | ON (profile) | 24 | **1** โ€” `acq_q12` | +| `ad_fire_esc_on` | `acqdocs` | OFF | 9 (the screened set) | **7** โ€” all but `acq_q07` and `docs_d15` | + +The `acq` disagreement (screen 2, agent 1) is decomposition: the agent retrieves +with the decomposer's rewrite, not the gold query, and the rewrite scores +differently. On `acqdocs` with decomposition off the screen and the agent still +disagree on 2 of 9, because the evidence-sufficiency retry's reformulation is +itself an LLM sample. + +**Fire subset, as executed: 1 query under the full shipped product default +(`acq`), 7 queries on the realistic corpus with decomposition off (`acqdocs`). +Both are below the nโ‰ˆ5 bar the brief set for the product-default case; the +7-query `acqdocs` set is right at it. This is an insufficient-n result and is +called as such in ยง7.** + +--- + +## 3. 4.1 A/B โ€” arm 1: shipped product default, `acq`, n=24 + +`document_escalation.enabled` false vs true. Everything else identical and at +the profile default, including query decomposition and verification. + +**Whole-set numbers** (context only โ€” 23 of 24 queries are byte-identical work +in both arms, so the whole-set figure is measuring generation nondeterminism, +not escalation): + +| arm | escalation events | majority-pass | judge run 1 | judge run 2 | mean pass fraction | mean synthesis `prompt_tokens` | +|---|---|---|---|---|---|---| +| `enabled: false` | 0 | 11/24 | 12/24 | 14/24 | 0.475 | 14563 | +| `enabled: true` | **1** | 10/24 | 12/24 | 10/24 | 0.433 | 14892 | + +**The fire subset, n = 1:** + +| query | arm | judge (5 runs) | first two runs | synthesis `prompt_tokens` | escalation event | +|---|---|---|---|---|---| +| `acq_q12` | off | **4/5** | `[True, True]` | 19545 | โ€” | +| `acq_q12` | on | **1/5** | `[False, True]` | 20686 | `05_financial_adjustments.pdf`, 2/2 chunks, ~470 approx_tokens, `dense_contrast=0.1073 < 0.12` | + +**Harm case, individually (the only fire in this arm):** `acq_q12` โ€” *"How many +people does StartupXYZ employ, and how are they split across functions?"*, gold +*"47 employees: 32 in engineering, 8 in sales, 7 in operations."* The baseline +answered it (4/5). The escalated arm did not (1/5). The escalated document, +`05_financial_adjustments.pdf`, does not contain the headcount โ€” the trigger +escalates *the top-ranked chunk's* document, and on this query the top-ranked +chunk is not in the answer-bearing document. n = 1; this is one sample of a +noisy process, not a demonstration. It is reported because the brief asks for +harm cases individually and because the mechanism it illustrates +(escalate-the-top-chunk's-document โ‰  escalate-the-right-document) recurs in +ยง5's 4.2 numbers. + +### 3.2 One non-escalation failure worth recording + +`acq_q04` in the escalation-on arm returned *"I could not find an answer in the +documents."* in 10.6s with **no synthesis call at all**. Cause, verbatim from +the pipeline log: + +``` +Decomposed into 1 sub-queries: ['"What proportion of the target company\'s turnover comes from its single biggest client?"'] +Could not search table 'w3b_acq': lance error: Invalid user input: position is not found + but required for phrase queries, try recreating the index with position, โ€ฆ +Could not search table 'w3b_acq_lc': lance error: Invalid user input: Cannot perform full + text search unless an INVERTED index has been created on at least one column, โ€ฆ +--- Final Documents for Synthesis --- +No documents to synthesize. +``` + +The decomposer emitted its single sub-query wrapped in double quotes; LanceDB +read that as a phrase query, the FTS leg raised, and **the whole hybrid +retrieval returned nothing** rather than degrading to the dense leg. Nothing to +do with 4.1 โ€” it happened to land in the escalation arm โ€” but it is a +retrieval-path bug that silently converts a normal query into "no answer", and +it is logged in ยง8. + +--- + +## 4. 4.1 A/B โ€” arm 2: `acqdocs`, fire subset, n=9 screened / 7 fired + +> **VOID โ€” gate note, 2026-08-12.** Every number in this section was measured +> under the ยง6 truncation bug and should not be cited: the 0/9 baseline below +> reads **7/9 on identical inputs** once prompts fit. The re-run this file's ยง9 +> demanded is `phase4-escalation-rerun.md`; its verdict (reject as default) +> supersedes this section. ยง3 (`acq`) is largely unaffected โ€” that corpus's +> prompts mostly fit even before the fix. + +Decomposition off in both arms (ยง1.3b) so the retrieval query is the gold query. +`escalation off` vs `escalation on`; nothing else differs. + +| query | judge OFF (5 runs) | first two | judge ON (5 runs) | first two | `prompt_tokens` off โ†’ on | escalation event (ON arm) | +|---|---|---|---|---|---|---| +| `acq_q04` | 0/5 | `[F, F]` | **2/5** | `[F, T]` | 8303 โ†’ 14732 | `02_due_diligence_report.pdf` 1/1 ch, ~540 tok, score 0.0915 | +| `acq_q07` | 0/5 | `[F, F]` | 0/5 | `[F, F]` | 8304 โ†’ 8304 | *did not fire* | +| `acq_q09` | 0/5 | `[F, F]` | **4/5** | `[T, T]` | 8311 โ†’ 8311 | `05_financial_adjustments.pdf` 2/2 ch, ~470 tok, score 0.0924 | +| `acq_q12` | 0/5 | `[F, F]` | 0/5 | `[F, F]` | 11242 โ†’ 13064 | `prompt_inventory.md` 16/16 ch, ~1797 tok, score 0.0566 | +| `docs_d03` | 0/5 | `[F, F]` | **4/5** | `[T, T]` | 8298 โ†’ 8298 | `triage_system.md` 10/10 ch, ~1768 tok, score 0.0926 | +| `docs_d05` | 0/5 | `[F, F]` | 0/5 | `[F, F]` | 8298 โ†’ 8298 | `indexing_pipeline.md` 47/47 ch, ~5460 tok, score 0.0644 | +| `docs_d13` | 0/5 | `[F, F]` | 0/5 | `[F, F]` | 8298 โ†’ 8298 | `api_reference.md` 58/58 ch, ~5436 tok, score 0.080 | +| `docs_d15` | 0/5 | `[F, F]` | **5/5** | `[T, T]` | 8302 โ†’ 8302 | *did not fire* | +| `docs_d20` | 0/5 | `[F, F]` | 2/5 | `[F, T]` | 8303 โ†’ 8303 | `retrieval_pipeline.md` 38/45 ch, ~6000 tok (**truncated at the budget**), score 0.0629 | + +| arm | majority-pass | judge run 1 | judge run 2 | mean pass fraction | +|---|---|---|---|---| +| escalation **off** | **0/9** | 0/9 | 0/9 | 0.000 | +| escalation **on** | **3/9** | 3/9 | 5/9 | 0.378 | + +**Restricted to the 7 queries that actually escalated: 0/7 โ†’ 2/7 majority-pass +(`acq_q09`, `docs_d03`).** + +**Harm cases: none.** No query passed in the baseline and failed in the +escalated arm โ€” but the baseline scored 0/9, so this arm structurally *cannot* +observe harm. That is a property of the subset (weak-evidence queries are the +ones the baseline gets wrong), not evidence that escalation is safe. + +**One of the three "help" rows is pure noise.** `docs_d15` went 0/5 โ†’ 5/5 and +**escalation did not fire on it** โ€” the two arms ran identical configurations. +That single row is a direct, in-sample measurement of how far generation +nondeterminism alone can move a query: the entire distance from certain-fail to +certain-pass. ยง5.2 measures the same thing on 11 rows. + +The two genuine helps are large and legible. `acq_q09`, baseline +(verbatim, truncated): + +> "Based on the provided text fragments, **there is no information** to answer +> your specific question โ€ฆ The documents you supplied describe: A RAG +> (Retrieval-Augmented Generation) system architecture (localGPT) โ€ฆ" + +and escalated: + +> "**First Disclosed Debt:** $1,500,000 ยท **Extra Borrowing Found Later:** +> $175,000 (identified as capital lease obligations)" + +against gold *"Due diligence disclosed $1.5 million of debt; a further $175,000 +of capital lease obligations was identified afterwards."* The baseline did not +give a wrong answer from the acquisition documents โ€” **it never saw them.** Why +that happens is ยง6, and it is the reason this lift cannot be read as a clean 4.1 +win. + +--- + +## 5. 4.2 A/B โ€” the cross-reference hop on the 11 `requires_crossref` rows + +Escalation off in every arm. Decomposition off in every arm. + +### 5.1 Primary: `acq`, shipped `retrieval_k = 20` โ€” the hop fires zero times + +| arm | queries that hopped | chunks added | mean synthesis `prompt_tokens` | majority-pass | judge run 1 | judge run 2 | +|---|---|---|---|---|---|---| +| `crossref_hop off` | 0/11 | 0 | 9369 | 4/11 | 6/11 | 4/11 | +| `crossref_hop on` | **0/11** | 0 | 9369 | 5/11 | 6/11 | 4/11 | + +Per-row `prompt_tokens` are **identical to the token** in both arms on all 11 +rows, which is the proof that the two arms fed the model the same context. + +**The mechanism, and it is not the one the previous wave found.** References now +resolve โ€” 34 of them, 9 of 10 documents linked (ยง1.1). The blocker is the +`_crossref_hop` "not already represented" guard. Read directly off a +retrieval-only run of all 11 rows: + +``` +acq_q13 cands=13 hopped=0 top3_refs=21 unrepresented_targets=0 +acq_q14 cands=13 hopped=0 top3_refs=23 unrepresented_targets=0 +โ€ฆ (identical shape on all 11 rows) +acq_q23 cands=13 hopped=0 top3_refs=14 unrepresented_targets=0 +fired 0/11 +``` + +The corpus is 13 chunks; at `k = 20` every query retrieves all 13, so all ten +documents are candidates; the top-3 candidates carry 14โ€“23 resolved references +and **every one of them points at a document that is already in the candidate +set**. The guard is behaving exactly as designed. There is nothing to fetch. + +The same probe on `acqdocs` at `k = 20` (373 chunks, so the guard can bite): +**2 of 11 rows hop** (`acq_q17` โ†’ `10_closing_checklist.pdf`, +`acq_q18` โ†’ `01_acquisition_agreement.pdf`), and **0 of those 2 landed in the +row's `expected_sources`.** + +### 5.2 That cell is also this file's noise floor + +Because ยง5.1's two arms provably fed the model identical context, every judged +difference between them is generation + judge noise. It is large: + +| query | judge OFF | judge ON | note | +|---|---|---|---| +| `acq_q13` | 0/5 | **5/5** | identical prompts, opposite verdicts | +| `acq_q15` | 2/5 | 0/5 | | +| `acq_q22` | 5/5 | 3/5 | | +| `acq_q19` | 0/5 | 2/5 | | +| aggregate | 4/11 majority, mean 0.418 | 5/11 majority, mean 0.509 | | + +**A one-row (0.09 mean-pass-fraction) aggregate difference on n = 11 is +indistinguishable from noise, and an individual row can swing the full 0/5 โ†’ +5/5.** Every judged delta elsewhere in this file has to clear that bar. + +### 5.3 Exploratory arm: `acq` at `retrieval_k = 5`, where the hop can fire + +Clearly labelled exploration. `k = 5` is not the shipped value; it is the +smallest `k` at which the "already represented" guard stops suppressing every +target on a 13-chunk corpus. Escalation off, decomposition off, both arms. + +| query | judge OFF | first two | judge ON | first two | `prompt_tokens` off โ†’ on | hop target (1 hop, `max_hops=1`) | row's `expected_sources` | hop hit expected? | +|---|---|---|---|---|---|---|---|---| +| `acq_q13` | 2/5 | `[F, T]` | **4/5** | `[T, T]` | 3981 โ†’ 4624 | `02_due_diligence_report.pdf` (1 ch) | `08_regulatory_approval.pdf` | no | +| `acq_q14` | 2/5 | `[T, F]` | **5/5** | `[T, T]` | 3966 โ†’ 4387 | `08_regulatory_approval.pdf` (1 ch) | `04_risk_assessment.pdf` | no | +| `acq_q15` | 4/5 | `[F, T]` | 5/5 | `[T, T]` | 4054 โ†’ 4853 | `10_closing_checklist.pdf` (2 ch) | `05_financial_adjustments.pdf` | no | +| `acq_q16` | 5/5 | `[T, T]` | 3/5 | `[T, T]` | 3595 โ†’ 4225 | `04_risk_assessment.pdf` (1 ch) | `03_ip_certification.pdf` | no | +| `acq_q17` | 0/5 | `[F, F]` | 0/5 | `[F, F]` | 3506 โ†’ 4347 | `01_acquisition_agreement.pdf` (2 ch) | `02_due_diligence_report.pdf` | no | +| `acq_q18` | 2/5 | `[F, F]` | 0/5 | `[F, F]` | 3670 โ†’ 4438 | `01_acquisition_agreement.pdf` (2 ch) | `03_ip_certification.pdf` | no | +| `acq_q19` | **4/5** | `[T, T]` | **0/5** | `[F, F]` | 3265 โ†’ 4106 | `01_acquisition_agreement.pdf` (2 ch) | `09_customer_consents.pdf` | no | +| `acq_q20` | 3/5 | `[F, T]` | 3/5 | `[T, T]` | 3764 โ†’ 4417 | `01_acquisition_agreement.pdf` (2 ch) | `08_regulatory_approval.pdf` | no | +| `acq_q21` | **5/5** | `[T, T]` | **2/5** | `[T, T]` | 3927 โ†’ 4936 | `02_due_diligence_report.pdf` (1 ch) | `05_financial_adjustments.pdf` | no | +| `acq_q22` | 0/5 | `[F, F]` | 1/5 | `[F, F]` | 3605 โ†’ 4446 | `01_acquisition_agreement.pdf` (2 ch) | `10_closing_checklist.pdf` | no | +| `acq_q23` | 1/5 | `[T, F]` | 0/5 | `[F, F]` | 3987 โ†’ 4580 | `01_acquisition_agreement.pdf` (2 ch) | `05_financial_adjustments.pdf`, `08_regulatory_approval.pdf` | no | + +| arm | rows that hopped | chunks added | hop hit an `expected_sources` document | majority-pass | judge run 1 | judge run 2 | mean pass fraction | mean `prompt_tokens` | +|---|---|---|---|---|---|---|---|---| +| hop **off** | 0/11 | 0 | โ€” | **5/11** | 5/11 | 6/11 | 0.509 | 3756 | +| hop **on** | **11/11** | 18 | **0/11** | **5/11** | 6/11 | 6/11 | 0.418 | 4487 | + +**Harm cases, individually:** + +* **`acq_q19`** (4/5 โ†’ 0/5) โ€” *"The legal opinion excepts change-of-control + provisions โ€ฆ For each affected customer, what is the current consent status?"* + Gold: MegaCorp obtained, DataFlow obtained, CloudTech pending. The answer is in + `09_customer_consents.pdf`. The hop pulled 2 chunks of + `01_acquisition_agreement.pdf` instead and displaced the row's own evidence + from a 5-candidate budget. Largest single regression in this file. +* **`acq_q21`** (5/5 โ†’ 2/5) โ€” *"The closing checklist calls for an escrow + agreement tied to Exhibit C. Over what period is that escrow released?"* Gold: + the $1,300,000 escrow releases over 18 months, in + `05_financial_adjustments.pdf`. The hop pulled `02_due_diligence_report.pdf`. +* **`acq_q16`** (5/5 โ†’ 3/5) and **`acq_q23`** (1/5 โ†’ 0/5) are smaller moves in + the same direction, inside the ยง5.2 noise band. + +Helps: `acq_q13` (2/5 โ†’ 4/5) and `acq_q14` (2/5 โ†’ 5/5) โ€” both with a hop that +did **not** hit the expected source, so if the hop helped there it helped by +accident. + +**Why the precision is 0/11.** `max_hops = 1`, and the target is the *first* +unrepresented resolved reference found by scanning the top-3 candidates in +order, then that candidate's references in text order. The acquisition corpus is +hub-and-spoke: `01_acquisition_agreement.pdf` and `02_due_diligence_report.pdf` +are referenced by nearly every other document and are named early in nearly +every chunk. So the one hop the budget allows is spent on the hub, essentially +every time, while the answer lives in a spoke. The gate's own prediction โ€” that +`crossref_hit_expected_source` would be the column to watch โ€” is confirmed, and +it reads zero. + +--- + +## 6. The finding that shadows every 4.1 number: the generation context does not fit + +`OLLAMA_CONTEXT_LENGTH=16384` on this host, and Ollama divides that across +parallel slots; with the second agent's traffic sharing the server, a standalone +probe measured the effective ceiling at **8194 prompt tokens**: + +``` +prompt 46072 chars -> prompt_eval_count 8038 +prompt 138072 chars -> prompt_eval_count 8194 +prompt 460073 chars -> prompt_eval_count 8194 (front of the prompt is discarded) +``` + +`RetrievalPipeline._synthesize_final_answer` does no client-side truncation โ€” it +formats the whole context into one prompt and posts it. On the `acqdocs` arms +the pipeline's own log reports the constructed context at **81 837 โ€“ 101 469 +characters** (โ‰ˆ20 000 โ€“ 25 000 tokens) *before* escalation appends anything, +against a slot ceiling of ~8 300. Every one of those synthesis calls was +served a **front-truncated** prompt. + +Three consequences, all of which the gate needs: + +1. **The `prompt_tokens` column in ยง4 is a ceiling, not a cost.** On five of the + seven fired rows the two arms report *identical* `prompt_tokens` + (8298 / 8298, 8311 / 8311, โ€ฆ) even though the escalated arm appended up to + 6 000 tokens. Escalation's true prompt-cost on this corpus was not measurable + with `token_usage`; the honest figure is the `approx_tokens` on the + `document_escalation` event (470 โ€“ 6 000 tokens, and `docs_d20` hit the + budget cap and reported `truncated: true`, 38 of 45 chunks). +2. **The measured 4.1 lift is confounded.** The escalated block is appended to + the *end* of the facts string, and front-truncation keeps the tail. So on an + over-long prompt, escalation does not merely "add a document" โ€” it + *guarantees the escalated document survives truncation while the top-ranked + retrieved chunks are discarded.* `acq_q09`'s baseline answer ("the documents + you supplied describe a RAG system architecture") is that failure in the + open: the acquisition chunks had been truncated away and only + `Documentation/*.md` text reached the model. The escalated arm answered + correctly because the answer-bearing document was at the tail. That is a real + improvement on this deployment, but it is **not** evidence for 4.1's stated + thesis (a whole document in original order beats similarity-ranked chunks). + The same lift would presumably come from simply reordering or trimming the + context. +3. **"Lost in the middle" could not be measured as designed.** The intended harm + check โ€” the escalated arm getting wrong what the baseline got right โ€” has one + observation (`acq_q12`, ยง3) because the `acqdocs` baseline scored 0/9 and had + nothing to lose. + +--- + +## 7. Caveats, stated plainly + +* **n is small everywhere.** 1 fired query under the shipped product default; + 7 fired queries on the realistic corpus; 11 rows for 4.2. One query is 0.14 of + the 4.1 fire subset and 0.09 of the 4.2 set. +* **The noise floor is ~1 row on n = 11 and up to a full 0/5 โ†’ 5/5 on a single + row** (ยง5.2, `acq_q13`; ยง4, `docs_d15`). Nothing in this file with a delta of + one or two rows is a result. +* **The judge is strict and, in this framing, noisy.** 20/20 on its own + validation set, but 5โ€“7 self-disagreements per 24 rows on real system answers, + which is why everything is `k/5`. It also produces defensible-but-harsh + negatives: `docs_d15`'s baseline said `.txt` files are "read directly into + fenced markdown without parsing or OCR" against gold "bypass docling and are + wrapped in a fenced code block" and was rejected 5/5 for not naming docling. + **Absolute pass rates in this file should not be compared to any other + document's.** Only the within-table paired direction means anything. +* **The judge measures gold-fact recall, not answer precision** (ยง1.4). +* **Wall clock is meaningless here** โ€” a second agent shared the GPU throughout, + and the same query ranged 11s to 272s across runs. +* **Two configuration deviations** (`think:false`; decomposition off in the + `acqdocs` and 4.2 arms) are symmetric across arms but mean these numbers do + not describe a byte-for-byte shipped run. +* **`chunk_size = 512`, not the profile's 1500.** Chosen so documents are + multi-chunk at all; at 1500 the acquisition PDFs are ~1 chunk each and 4.1 is a + no-op by construction. This makes the corpus *more* favourable to 4.1 than the + shipped chunking would be. +* **`atlas7` / `hr` were not run** (ยง1.1) โ€” 1 and 2 chunks. +* **4.3 (overview prefilter) was not touched.** Out of scope for this brief. +* **The `acq` corpus is 13 chunks.** At the shipped `k = 20` no + candidate-selection change can move anything on it; ยง5.1 is the clean + demonstration of that, not a measurement of the hop. + +--- + +## 8. Backlog this wave created + +* **The 9b generation model returns an empty answer under the shipped + defaults.** `stream_completion` does not pass `think`, so the model thinks + until the context is exhausted and emits no `response`. Reproduced verbatim in + ยง1.3a (`9351 + 7033 = 16384`, answer = `""`). The repo already defaults + thinking off for `format="json"` calls; the synthesis path does not. This is a + product bug, not an eval artifact, and it is the highest-value item here. +* **The synthesis prompt is not budgeted against the generation model's context + window.** ยง6: 20 000 โ€“ 25 000 token prompts posted into an 8 200-token slot, + silently front-truncated, so the highest-ranked retrieved chunks are the first + thing thrown away. Any context-window guard (`retrieval_k` cap by token budget, + or a client-side trim that drops from the *bottom* of the ranking) would change + the answers on this corpus more than either Phase-4 feature does. +* **A quoted sub-query from the decomposer kills hybrid retrieval outright.** + ยง3.2: LanceDB reads `"โ€ฆ"` as a phrase query, the FTS leg raises, and + `retrieve()` returns nothing instead of degrading to the dense leg. One query + in 24 in this session. +* **`crossref_hop` picks its target by scan order, not by relevance.** ยง5.3: + 0/11 document-level precision on a hub-and-spoke corpus because the one + permitted hop is always spent on the hub. If 4.2 is ever revisited, ranking + candidate targets (by the query's similarity to the target document's + overview, for instance) is the change that would matter โ€” `max_hops` is not. + +--- + +## 9. Proposed calls, with the one line each rests on + +* **4.1 โ€” HOLD.** 0/7 โ†’ 2/7 on the fire subset is directionally positive and + larger than the noise floor, but ยง6 shows the mechanism producing it is prompt + truncation rather than document reassembly, and the single fire under the true + product default was a regression. Fix the context-window bug in ยง8 first, then + re-run this exact A/B; if the lift survives a prompt that actually fits, adopt. + Do not turn the flag on now, and do not delete the code. +* **4.2 โ€” REJECT as a shipped default; keep the flag.** At the shipped + `retrieval_k = 20` it fires 0/11 on the rows built for it (ยง5.1) โ€” so turning + it on buys nothing โ€” and the only configuration where it does fire lands on the + wrong document 11/11 and is a judged wash with two clear harms (ยง5.3). + Index-time extraction (`indexing.extract_crossrefs`, already ON by default) is + unaffected by this call and should stay on: it is free, it is correct after the + gate's resolver fix (34/68 resolved on `acq`), and it is what a future, + relevance-ranked hop would need. diff --git a/eval/decisions/phase4-crossref-prefilter.md b/eval/decisions/phase4-crossref-prefilter.md new file mode 100644 index 00000000..bb5d2057 --- /dev/null +++ b/eval/decisions/phase4-crossref-prefilter.md @@ -0,0 +1,395 @@ +# Phase 4 items 4.2 and 4.3 โ€” cross-reference hop and overview prefilter + +Date: 2026-08-09 +Status: **implemented, flag-gated, both defaults OFF, not yet benchmarked.** +Owner of this wave: retrieval/indexing pipelines only. `main.py` and +`Documentation/*.md` were deliberately **not** edited โ€” the config keys and doc +diffs this change needs are *proposed* at the bottom of this file and belong to +the adoption gate. + +There is **no gold-set evidence for or against either feature yet**. The mixed +gold set has no cross-reference queries and no multi-document overview queries +(roadmap Phase 4: "4.2 and 4.3 need multi-document gold queries with +cross-references added to eval/goldset"). Everything below is a mechanism proof +plus a regression proof that the mechanisms are inert while switched off. Do not +turn either flag on in a shipped profile on the strength of this document. + +--- + +## 1. What shipped + +### 4.2 Cross-reference hop + +**Index time** โ€” `rag_system/indexing/crossref.py` (new), +`rag_system/pipelines/indexing_pipeline.py`. + +Deterministic regexes over each chunk's *original* text (the pass runs after +chunking and **before** contextual enrichment, so an LLM-written preamble can +neither invent nor swallow a reference). Three reference families: + +| kind | matched | example `ref` | +|------|---------|---------------| +| `exhibit` | `exhibit / appendix / schedule / annex / attachment / addendum` + a single capital letter or a dotted number, optional `No.` / `#` | `exhibit b`, `schedule 2.1`, `appendix a` | +| `section` | `section / clause / article` + a dotted number, and the `ยง` symbol form | `section 4.3`, `section 7` | +| `document` | the normalized filename/title of *another* document in the resolution set, appearing as whole words in the chunk | `northwind leave policy` | + +Stored as `metadata.crossrefs = [{"kind", "ref", "target_doc"}]`. Capped at 8 +distinct references per chunk (more than that means a table of contents). +Chunks with no references get **no key at all**, so metadata does not grow for +corpora that have none. + +Resolution is name-based only: a reference resolves when some document's +normalized filename/title contains it as whole words (`"Exhibit B"` โ†’ +`exhibit_b.pdf`). The resolution set is the current indexing batch **plus** the +document ids already in the target LanceDB table, so an incremental add can +still point at a document indexed last week (best effort; any failure reading +the table silently falls back to batch-only with one log line). Unresolvable +references are still recorded with `target_doc: null` โ€” they are true, a UI can +show them, they just have nowhere to hop to. + +**Never resolves to the chunk's own document.** Otherwise `exhibit_b.pdf` saying +"this Exhibit B" self-resolves on every chunk, and a document that repeats its +own title mints a useless reference per chunk. + +Config: `indexing.extract_crossrefs`, default **TRUE**. Extraction is a handful +of regexes over text already in memory โ€” no LLM, no second pass โ€” and it only +writes chunk metadata. The `text` and `vector` columns are unchanged, which is +why it is safe on by default while the query-time hop is not. + +**Query time** โ€” `rag_system/pipelines/retrieval_pipeline.py`. + +`retrieval.crossref_hop = {"enabled": false, "max_hops": 1, "chunks_per_hop": 3}`. + +After the candidate set is final (first stage + rerank + the evidence-sufficiency +retry), each of the **top 3** candidates is inspected for a `crossrefs` entry +whose `target_doc` is a document **not already represented** in the candidates. +Up to `max_hops` such documents are expanded: a dense, document-filtered LanceDB +search pulls that document's `chunks_per_hop` most on-topic chunks, and they are +appended to `documents` tagged `via_crossref: true` (both top-level and inside +`metadata`, plus a `crossref` record naming the source chunk and the reference). + +Bounds, all hard: only the top 3 candidates can trigger a hop; hopped chunks can +never trigger another one (no recursion); no LLM anywhere in the path; the +worst case is `max_hops` extra filtered vector searches per query. + +`result["crossref_hop"]` carries the hop record, and an event `crossref_hop` is +emitted so the UI/citations can show it. `first_stage` is **never** mutated โ€” +with reranking off, `documents` and `first_stage` are the same list object, so +the hop rebuilds `documents` as a new list. + +Two downstream interactions were fixed in `run()`: + +* Context expansion re-reads the row from LanceDB, which knows nothing about how + the chunk arrived โ€” the `via_crossref` marker is now carried over onto the + central chunk the same way `rerank_score` already was. +* The "hide non-reranked chunks" filter would otherwise delete every hopped + chunk whenever the reranker is on, since hops are appended *after* reranking + by design. Hopped chunks are now exempt from that filter. (Scoring them with + the reranker would defeat the point: the referenced document is precisely the + one whose text does not look like the query.) + +### 4.3 Overview prefilter + +**Index time** โ€” `rag_system/indexing/overview_builder.py`, +`rag_system/pipelines/indexing_pipeline.py`. + +At the end of an index build, every overview in +`index_store/overviews/<index_id>.jsonl` is embedded with the **document-side** +embedder (no instruction prefix โ€” same asymmetry as the chunk index) and written +to a sidecar: + +``` +index_store/overviews/<index_id>.jsonl the overviews (unchanged) +index_store/overviews/<index_id>.vectors.npz doc_ids + L2-normalized vectors + {embedding_model, normalized} +``` + +Rebuilt wholesale rather than appended, because the JSONL is append-only and a +re-indexed document has several lines of which only the last is current. Cost is +one embedding per *document*. Failures print a warning and never fail the build. +Config: `overview.embed`, default TRUE (only reachable when `overview.enabled`). + +It is a sidecar and not a LanceDB table because it is one row per document (tens, +not thousands), it is rebuilt wholesale rather than queried, and a missing +sidecar has to be a graceful no-op rather than a schema problem. + +**Query time** โ€” +`retrieval.overview_prefilter = {"enabled": false, "top_documents": 5, "mode": "boost"}`. + +The query vector is scored against the overview vectors (cosine; both sides +L2-normalized) and the top `top_documents` documents are selected. Then: + +* `mode: "boost"` (the default when enabled โ€” the safer of the two): the + candidate ordering is fused with the document-overview ordering by **RRF at the + same `_RRF_K = 60` the retriever uses**. A rank bonus, not a score bonus, and + no weight knob โ€” for the reason design_rationale ยง4 gives for the BM25/dense + fusion: the two orderings are not on a common scale and there is no validation + split here to tune a weight against. Nothing is dropped; documents outside the + top-N simply contribute no second leg. +* `mode: "restrict"`: the first stage itself runs with a LanceDB + `document_id IN (โ€ฆ)` prefilter. If the restricted search returns nothing, it + logs and falls back to unrestricted retrieval for that query. + +Both are computed **inside `_first_stage`**, not around it, so the +evidence-sufficiency retry re-scores documents against its reformulated query +too. + +Sidecar path resolution, most explicit first: +`retrieval.overview_prefilter.vectors_path` โ†’ `config["overview_path"]` with +`.jsonl` swapped for `.vectors.npz` โ†’ `index_store/overviews/<index_id>.vectors.npz`. +Nothing beyond that is guessed: silently prefiltering against *some other +index's* overviews would be worse than not prefiltering. `api_server.py` already +sets `rp_cfg["overview_path"]` per session (`:367`), so the HTTP path needs no +new plumbing for resolution โ€” only the flag. + +**Graceful degradation** (all one log line, then normal retrieval): no overview +path configured; sidecar file absent; sidecar unreadable; sidecar written by a +different embedding model than the pipeline is configured for. + +### Shared primitive + +`RetrievalPipeline._search_within_documents()` โ€” a document-filtered search that +mirrors `MultiVectorRetriever.retrieve` exactly (same prefiltered FTS and vector +legs, same RRF at `_RRF_K = 60`, same output row shape). It lives in the +retrieval pipeline rather than in `retrievers.py` because `retrieve()` has no +filter parameter and that module is owned elsewhere this wave. **If a filtered +variant is ever added to `MultiVectorRetriever`, this helper should be deleted +in favour of it** โ€” it is duplication, and it is only justified by the ownership +boundary. + +`retrieve_candidates` returns from five places (the retry has four early outs). +All five now route through `_post_candidates()`, a tail hook that runs whatever +must see a *final* candidate set. If another change needs the same, add it there +rather than to the five return sites. + +--- + +## 2. Verification + +### `py_compile` + +Clean on `rag_system/indexing/crossref.py`, +`rag_system/indexing/overview_builder.py`, +`rag_system/pipelines/indexing_pipeline.py`, +`rag_system/pipelines/retrieval_pipeline.py`. + +### Regression, both flags OFF โ€” `--corpus mixed` + +Same LanceDB index in every row below (built 2026-08-09T20:45:43Z, 363 chunks, +15 files), so these are code-to-code comparisons. + +| run | retry | R@5 | R@10 | R@20 | nDCG@10 (1st) | +|-----|-------|-----|------|------|---------------| +| **before** this change (`phase4_i2_before.json`) | on (profile) | 0.944 | 0.972 | 1.000 | **0.898** | +| after, run 1 (`phase4_i2_regression.json`) | on (profile) | 0.944 | 0.972 | 1.000 | **0.903** | +| after, run 2 (`phase4_i2_regression_run2.json`) | on (profile) | 0.958 | 0.986 | 1.000 | **0.887** | +| after, deterministic arm (`phase4_i2_regression_retryoff.json`) | off (forced) | 0.944 | 0.972 | 1.000 | **0.887** | +| Phase 2 gate, for reference (`gate_phase2_all.json`, 331 chunks / 14 files) | on (profile) | 0.944 | 0.972 | 1.000 | 0.9063 | + +In the 0.90โ€“0.91 band, and recall is bit-identical to the pre-change run. The +0.887โ€“0.903 spread across the retry-on runs is the evidence-sufficiency retry's +LLM reformulation, which is nondeterministic; the retry-off arm is deterministic +and lands at 0.887, against 0.8881 for the last recorded retry-off run +(`phase2_final_retry_off.json`) on a corpus that has since grown from 331 to 363 +chunks. **Honest caveat: I do not have a pre-change retry-off number on this +exact index, so the deterministic arm is compared against a different corpus +revision.** The code-level argument is the stronger one: with both flags absent, +`_overview_prefilter_documents` returns at the `enabled` check before any I/O and +`_crossref_hop` returns at the `enabled` check before touching the result, so +nothing on the flags-off path executes a new line. + +### Index-side inertness (`indexing.extract_crossrefs` defaults ON) + +The riskiest default in this change, so it was checked rather than argued. The +`mixed` corpus was rebuilt into a throwaway directory with the current pipeline +(extraction on) and compared chunk-by-chunk against the shared eval index, which +was built *before* this change: + +``` +old (pre-change build): 363 chunks +new (extraction ON) : 363 chunks +chunk_id sets identical: True +text column identical : True +vector shapes : (363, 1024) (363, 1024) +vectors bit-identical : True +max abs vector delta : 0.0 + +new index: 147 crossrefs across 59 chunks, 59 resolved +old index: 0 chunks carrying crossrefs (expected 0) +``` + +Extraction adds 147 references across 59 of 363 chunks on the real +`Documentation/*.md` corpus, 59 of them resolving to 12 documents, and moves +neither a byte of `text` nor a bit of any vector. + +### End-to-end smoke + +`.venv/bin/python eval/smoke_e2e.py` โ€” **25/25 assertions passed** (676.9s, +exit 0). Worth noting: the smoke's teardown reported removing a leaked +`index_store/overviews/<id>.vectors.npz` alongside the `.jsonl`, which +independently confirms the sidecar is produced on the real HTTP index-build path +and that the smoke's cleanup already globs it โ€” no change to `eval/` was needed. + +### Scratch functional test + +Not in `eval/` โ€” a throwaway 3-document index in a temp directory +(`master_agreement.md` references "Exhibit B" and "Schedule 2.1" and contains no +prices; `exhibit_b.md` is the rate card; `hr_handbook.md` is an unrelated +distractor). Full verbatim output is in the wave report. Deviation from the +brief: the documents are `.md`, not `.pdf`, because no PDF writer is installed in +this venv. The extension is stripped by name normalization, so the mechanism is +identical. + +1. **Extraction.** `master_agreement.md#0` and `#1` both carry + `{"kind": "exhibit", "ref": "exhibit b", "target_doc": "exhibit_b.md"}`; + `schedule 2.1` and the five `section N` references are recorded with + `target_doc: null`; `exhibit_b.md#0`'s own "Exhibit B" resolves to `null` + (self-reference suppressed). +2. **Hop, negative case** (`retrieval_k=3`, exhibit_b already a candidate): no + hop fires. The "not already represented" guard works. +3. **Hop, positive case** (`retrieval_k=1`, exhibit_b outside the candidates): + the single candidate is `master_agreement.md#0`; the hop pulls both + `exhibit_b.md` chunks tagged `via_crossref`, and the answer string `4,250` + enters the context. `first_stage` is unchanged. + *Caveat: `k=1` is induced. The whole corpus is 5 chunks, so at any k โ‰ฅ 2 the + referenced document is already a candidate. On a real corpus the situation is + the normal one; here it had to be forced.* +4. **Overview sidecar.** Written with 3 doc_ids, `(3, 1024)` vectors, + `meta={'embedding_model': 'microsoft/harrier-oss-v1-0.6b', 'normalized': True}`. +5. **Boost.** For *"How much does the Client have to pay to onboard a new + location?"* the first stage ranks `master_agreement.md` first and + `exhibit_b.md` second; the prefilter selects `exhibit_b.md` from its overview + ("a **rate card** outlining service pricingโ€ฆ") and boost flips the top-1 to + `exhibit_b.md`, which is the document holding the answer. +6. **Restrict.** Confines retrieval to `['exhibit_b.md']`. +7. **Degradation.** A missing sidecar prints one line and returns the identical + unfiltered result set. + +--- + +## 3. Proposed config keys (for `rag_system/main.py`, at the adoption gate) + +`PIPELINE_CONFIGS["default"]["retrieval"]`: + +```python + # Cross-reference hop (roadmap 4.2). OFF: index-time extraction is + # free and additive, but the query-time hop appends chunks the + # retriever never scored, and no gold query exercises it yet. + "crossref_hop": { + "enabled": False, + "max_hops": 1, # referenced documents expanded, no recursion + "chunks_per_hop": 3 + }, + # Overview prefilter (roadmap 4.3). OFF until benchmarked. "boost" + # is the safe mode โ€” it reorders; "restrict" can hide a document. + "overview_prefilter": { + "enabled": False, + "top_documents": 5, + "mode": "boost" # "boost" | "restrict" + } +``` + +`PIPELINE_CONFIGS["default"]["indexing"]`: + +```python + "extract_crossrefs": True # regex-only, no LLM; writes chunk metadata +``` + +`PIPELINE_CONFIGS["fast"]`: same two `retrieval` blocks with +`"enabled": False`, and `"extract_crossrefs": True` in `indexing` (it costs +nothing). `overview.embed` defaults to `True` in code and needs no profile entry +unless someone wants it discoverable. + +Both blocks are read through the same `retrieval`/`retrievers` merge the retry +and late-chunk blocks use, so an API runtime override written under +`retrievers.crossref_hop` / `retrievers.overview_prefilter` already wins over the +profile with no extra code. Wiring UI toggles in `api_server.py` is therefore a +two-line mapping per flag โ€” also a gate task, not this wave. + +## 4. Proposed documentation diffs (none applied this wave) + +* **`Documentation/indexing_pipeline.md`** + * New section *Cross-reference extraction*, between *Chunking* and *Document + overviews*: what the three regex families match, the `metadata.crossrefs` + shape, the "resolves against batch + existing table, never against itself" + rule, and the placement before contextual enrichment. + * *Document overviews*: add the `.vectors.npz` sidecar โ€” written at the end of + the build with the document-side embedder, rebuilt wholesale, one row per + document. + * *Pipeline config keys* table: add `indexing.extract_crossrefs` (default + `true`) and `overview.embed` (default `true`). +* **`Documentation/retrieval_pipeline.md`** + * New stages for the cross-reference hop (after the retry, before context + expansion) and the overview prefilter (inside the first stage), both marked + default-off; the `crossref_hop` SSE event; the `via_crossref` /`crossref` + fields on a source document. +* **`Documentation/design_rationale.md`** + * ยง2 (chunking + index-time enrichment): one paragraph on why cross-reference + extraction is index-time regex rather than an LLM pass, and why it precedes + enrichment. + * ยง4 (hybrid retrieval + RRF): note that the overview-prefilter boost reuses + RRF at the same `_RRF_K` and deliberately introduces no weight, consistent + with "there are no fusion weights, and there is no knob to add them". + * A new short section (or ยง13 entry) recording that both features are shipped + dark pending gold queries โ€” the roadmap's own precondition. +* **`Documentation/research_roadmap.md`** + * Phase 4 table rows 4.2 and 4.3: mark implemented-but-gated, pointing here. + +## 5. What would settle it + +Gold queries. Specifically: + +* **4.2** โ€” at least 4โ€“6 `mixed` rows whose answer text lives in a document that + is *only* reachable through a reference in another document, with `expected` + anchored on the referenced document's text. Today's gold set cannot move at + all when the hop is on, because the hop only ever *appends*, and the harness + measures `first_stage`, which the hop never touches. **The harness will + therefore report a flat line for 4.2 no matter how well it works** โ€” measuring + it needs either a post-hop metric or gold rows scored on `documents`. +* **4.3** โ€” multi-document rows where the answer-bearing document is not the + lexically closest one. `boost` and `restrict` should be run as separate arms; + `restrict` also needs a *harm* check, because it can remove the correct + document from consideration entirely, and recall@20 is where that would show. +* Both need the `mixed` corpus to actually contain a cross-referenced document + pair; it does not today. + +--- + +## Gate correction (2026-08-09) โ€” resolver could not fire on the target corpus + +The first 4.2 A/B (eval/decisions/phase4-eval-final-metric.md) measured **zero +hops on `acq`** and 7 non-gold hops on `acq+docs`. Root cause, verified at the +gate directly against the built indexes: all 34 references extracted from the +acquisition corpus had `target_doc: null`. Two reasons: + +1. Exhibits/Schedules in this corpus are *sections inside* + `01_acquisition_agreement.pdf`, not separate files โ€” unresolvable by design, + and correctly left null. +2. Document-name mentions could never match, because `normalize_name` keeps the + numeric filename prefix (`08_regulatory_approval.pdf` โ†’ `"08 regulatory + approval"`), and prose says "the Regulatory Approval documentation", never + "08 regulatory approval". + +**Fix applied at the gate** (`rag_system/indexing/crossref.py`, +`CrossRefExtractor.__init__`): each known document is additionally registered +under its numeric-prefix-stripped name (`"regulatory approval"`), subject to the +same `_MIN_NAME_CHARS`/`_MIN_NAME_TOKENS` guards. The full-name entry wins ties; +self-suppression is unchanged (it compares resolved doc ids, not names). + +Verified at the gate on the real corpus text: `acq` extraction went from +0/34 resolved to **34/68 resolved, 9 of 10 documents linked**, with sensible +edges (due_diligence_report โ†’ regulatory_approval, risk_assessment โ†’ +closing_checklist, โ€ฆ) and no self-resolution. Original behaviors regression- +tested (exhibit_b resolution, self-suppression, label-pass dedup). + +Consequences for the Wave-3 re-measurement: + +* The `acq` and `acq_plus_docs` eval indexes must be **rebuilt** before any hop + arm runs โ€” crossrefs are stamped at index time and the existing indexes + predate both the extractor and this fix. +* Known laxness accepted: a stripped single-word alias ("01_overview.pdf" โ†’ + "overview") can over-match; the guards limit but do not eliminate this. The + hop A/B's `crossref_hit_expected_source` precision column is the check. +* At the product's `retrieval_k=20`, appended hop chunks sit beyond rank 10 and + cannot move nDCG@10 by construction. The meaningful mechanism test is small-k + (`--k 3`, `--k 5`) recall/nDCG on the final list; the meaningful product test + is judged end-to-end answers on the `requires_crossref` rows, hop on vs off. diff --git a/eval/decisions/phase4-escalation-rerun.md b/eval/decisions/phase4-escalation-rerun.md new file mode 100644 index 00000000..be81d41b --- /dev/null +++ b/eval/decisions/phase4-escalation-rerun.md @@ -0,0 +1,366 @@ +# Roadmap 4.1 (full-document escalation) โ€” re-run after the context-window fix + +Date: **2026-08-12**. Branch `rearchitect/evidence-gated-aug-2026`, HEAD **007d0b6**, +interpreter `.venv/bin/python`, Ollama `localhost:11434`, generation `qwen3.5:9b`, +judge/enrichment `qwen3.5:4b`. + +This re-runs the exact A/B that `eval/decisions/phase4-answer-quality.md` ยง9 put on +HOLD, under the condition that file set: *"Fix the context-window bug in ยง8 first, +then re-run this exact A/B; if the lift survives a prompt that actually fits, adopt."* + +**Answer: the lift does not survive. It was the truncation artifact ยง6 predicted it +might be.** Proposed call in ยง7 below. + +No file under the repo was edited by this wave. Everything below lives in +`SCRATCH = /private/tmp/claude-501/-Users-prompt-videos-localgpt-08082026/4d62420b-7ab2-4be1-90f2-708d7bae9146/scratchpad`. + +--- + +## 1. Setup verification + +### 1.1 The fix is active + +`SCRATCH/verify_num_ctx.py`, the 99k-char probe that previously returned +`prompt_eval_count = 8194` and lost a fact planted at character 0: + +``` +sizing helper: OK +prompt chars: 98763 +prompt_eval_count = 17351 (was 8194 before the fix) +eval_count = 9 +answer = 'ZEBRA-7741' + +context window fixed: PASS +front fact recovered: PASS +``` + +`/api/ps` during the runs reports `qwen3.5:9b` loaded at `context_length 32768`. + +### 1.2 Indexes โ€” opened, not rebuilt + +``` +w3b_acq 13 chunks w3b_acq_lc 13 +w3b_acqdocs 373 chunks w3b_acqdocs_lc 373 +``` + +Matches the 13 / 373 recorded on 2026-08-09. The 2026-08-09 result files in +`w3b_indexes/results/*.jsonl` were read only; every new file carries the `_fix` +suffix. + +### 1.3 The only code change since the 2026-08-09 runs is the fix itself + +``` +007d0b6 2026-08-12 13:05 fix: size Ollama num_ctx per request +c2d9f48 2026-08-09 18:02 docs: Phase-4 verdicts โ€ฆ (Documentation/*.md ONLY โ€” 4 files, no rag_system/) +e3c3d60 2026-08-09 18:02 eval: acquisition corpus โ€ฆ +``` + +The 2026-08-09 arms finished at ~17:34, before `c2d9f48`; `c2d9f48` touches no +`rag_system/` file. The `acqdocs` index was built at 16:24 that day and was **not** +rebuilt, so both dates retrieve byte-identical chunks. Attribution of any change +below to the num_ctx fix is therefore clean. + +### 1.4 Harness โ€” reused unmodified + +`w3b_common.py`, `w3b_run.py`, `w3b_judge.py` used as-is, same flags as +`chain1.sh`/`chain3.sh`. Two additions, both observation-only: + +* `w3b_ctxprobe.py` โ€” wraps `_warn_if_truncated` (the one hook that already sees both + the request payload and the final response on all three completion paths) and + `Agent.run`, to log per call: model, stage, prompt chars, requested `num_ctx`, + served `prompt_eval_count`, and whether the warning condition fired. Changes no + behaviour. +* `w3b_run_fix.py` โ€” imports `w3b_run`, installs the probe, calls `w3b_run.main()`. + +`think:false` on the generation model is still monkeypatched in the scratch runner +(the synthesis-path thinking bug is unfixed); symmetric across arms, as before. + +--- + +## 2. Prompts now fit โ€” the direct evidence + +Across all four arms: **249 completion calls, 0 truncation warnings.** + +| arm | calls | warned | max single-call `prompt_eval_count` | max `num_ctx` requested | +|---|---|---|---|---| +| `ad_fire_esc_off_fix` | 27 | **0** | 20 831 | 32768 | +| `ad_fire_esc_on_fix` | 27 | **0** | 27 516 | 32768 | +| `acq_esc_off_fix` | 102 | **0** | 9 365 | 16384 | +| `acq_esc_on_fix` | 93 | **0** | 9 966 | 32768 | + +9b synthesis calls that exceeded the old 8194 ceiling: **9/9** in each `acqdocs` arm, +36/47 and 33/42 in the `acq` arms. Representative rows (`acqdocs`, escalation ON): + +``` +docs_d20 chars=120272 num_ctx=32768 prompt_eval_count=27516 warned=False +docs_d13 chars=110102 num_ctx=32768 prompt_eval_count=26352 warned=False +docs_d05 chars=104394 num_ctx=32768 prompt_eval_count=23512 warned=False +acq_q04 chars= 95682 num_ctx=32768 prompt_eval_count=22059 warned=False +``` + +**No row filled its 32768 window** (highest 27 516, i.e. 84% of the window). There is +therefore **no residual-truncation subset to report separately** โ€” the confound is +gone for these arms, not merely reduced. + +### 2.1 What this changed vs 2026-08-09, per corpus + +`token_usage.by_stage.synthesis.prompt_tokens` is a **sum over 2 calls**, so compare +it only to itself: + +| corpus | 2026-08-09 `prompt_tokens` (esc OFF) | 2026-08-12 (esc OFF) | +|---|---|---| +| `acqdocs` (373 chunks) | 8298 โ€“ 8311 on 8 of 9 rows โ€” pinned at the ceiling | 11 242 โ€“ 20 940 | +| `acq` (13 chunks) | 9351 โ€“ 29 700 | 9 351 โ€“ 30 298 | + +This is the key asymmetry, and it explains everything downstream. The `acq` corpus +produces ~9.3k-token synthesis calls, which fit the old server window whenever the +box was uncontended โ€” 6 of its 24 rows report **byte-identical** `prompt_tokens` on +both dates (`acq_q01` 9353, `acq_q02` 9357, `acq_q03` 9351, `acq_q04` 9465, `acq_q05` +9357, `acq_q24` 9356), which cannot happen to a prompt being clipped to a contended +slot. The `acqdocs` corpus +produces 14kโ€“27k-token calls, which never fit, and 8 of its 9 rows sat exactly on the +8194 ceiling. **The 2026-08-09 `acqdocs` baseline was measuring a crippled system; +the `acq` baseline mostly was not.** + +--- + +## 3. Fire sets โ€” new vs old + +| cell | 2026-08-09 fire set | 2026-08-12 fire set | delta | +|---|---|---|---| +| `acqdocs` fire subset (decomp OFF) | `acq_q04, acq_q09, acq_q12, docs_d03, docs_d05, docs_d13, docs_d20` (7/9) | `acq_q04, acq_q07, acq_q09, acq_q12, docs_d03, docs_d05, docs_d13` (7/9) | **+`acq_q07`, โˆ’`docs_d20`** | +| `acq` product default (decomp ON) | `acq_q12` (1/24) | `acq_q04, acq_q12` (2/24) | **+`acq_q04`** | + +Same size on the deciding cell, one member swapped, exactly as the brief anticipated +(the evidence-sufficiency retry's reformulation is a fresh LLM sample each run). All +escalation events fired on `dense_contrast` below the 0.12 threshold. **No event in +either new arm hit the 6000-token budget cap** (`truncated: false` on all 9), whereas +2026-08-09's `docs_d20` reported `truncated: true` at 38/45 chunks. + +--- + +## 4. CELL 1 (deciding) โ€” `acqdocs` fire subset, decomposition OFF, n=9 + +`k/5` votes, majority โ‰ฅ3/5. `OLD` columns are the 2026-08-09 files, recomputed from +`judge5_ad_fire_esc_*.jsonl`, not copied from prose. + +| qid | fired | **OFF k/5** | **ON k/5** | dir | OLD OFF | OLD ON | old dir | pt offโ†’on | OLD pt offโ†’on | +|---|---|---|---|---|---|---|---|---|---| +| `acq_q04` | Y | **4/5** | **0/5** | HARM | 0/5 | 2/5 | = | 20940 โ†’ 22168 | 8303 โ†’ 14732 | +| `acq_q07` | Y | **5/5** | **5/5** | = | 0/5 | 0/5 | = | 17134 โ†’ 17535 | 8304 โ†’ 8304 | +| `acq_q09` | Y | **4/5** | **0/5** | HARM | 0/5 | 4/5 | HELP | 14586 โ†’ 21428 | 8311 โ†’ 8311 | +| `acq_q12` | Y | 0/5 | 0/5 | = | 0/5 | 0/5 | = | 11242 โ†’ 13064 | 11242 โ†’ 13064 | +| `docs_d03` | Y | **4/5** | **5/5** | = | 0/5 | 4/5 | HELP | 20384 โ†’ 22216 | 8298 โ†’ 8298 | +| `docs_d05` | Y | **4/5** | 2/5 | HARM | 0/5 | 0/5 | = | 19383 โ†’ 23616 | 8298 โ†’ 8298 | +| `docs_d13` | Y | 1/5 | 0/5 | = | 0/5 | 0/5 | = | 16941 โ†’ 26456 | 8298 โ†’ 8298 | +| `docs_d15` | . | **4/5** | **5/5** | = | 0/5 | 5/5 | HELP | 20412 โ†’ 24684 | 8302 โ†’ 8302 | +| `docs_d20` | . | **4/5** | **4/5** | = | 0/5 | 2/5 | = | 19934 โ†’ 27625 | 8303 โ†’ 8303 | + +**Majority-pass tallies:** + +| subset | n | OFF | ON | 2026-08-09 OFF | 2026-08-09 ON | +|---|---|---|---|---|---| +| all rows | 9 | **7/9** | 4/9 | **0/9** | 3/9 | +| new fired rows | 7 | **5/7** | **2/7** | โ€” | โ€” | +| old fired rows | 7 | 5/7 | 2/7 | **0/7** | **2/7** | +| union of fire sets | 8 | 6/8 | 3/8 | โ€” | โ€” | + +**The headline number.** The escalation-**off** baseline on this subset went from +**0/9 to 7/9** with no change other than the prompt fitting. The escalation-**on** +arm went from 3/9 to 4/9. The 2026-08-09 gap that motivated 4.1 (0/7 โ†’ 2/7) is gone; +the same rows now read **5/7 โ†’ 2/7**, i.e. the sign has flipped. + +`acq_q09` is the clearest single illustration, because ยง4 of the decision file quoted +it as the poster child. 2026-08-09: baseline 0/5 ("the documents you supplied describe +a RAG system architecture" โ€” it never saw the acquisition PDFs), escalated 4/5. +2026-08-12: **baseline 4/5**, answering from the retrieved chunks directly โ€” + +> "**Previously Disclosed Debt**: โ€ฆ **$1,500,000**. **Extra Borrowing Found Later**: +> โ€ฆ an additional **$175,000** in previously undisclosed capital lease obligations." + +โ€” and the escalated arm 0/5. The lift was the truncation, exactly as ยง6.2 warned. + +--- + +## 5. CELL 2 (product default) โ€” `acq`, n=24, decomposition ON + +| qid | fired | OFF k/5 | ON k/5 | dir | OLD OFF | OLD ON | old dir | +|---|---|---|---|---|---|---|---| +| `acq_q04` | **Y** | **4/5** | **0/5** | HARM | 0/5 | 0/5 | = | +| `acq_q12` | **Y** | **4/5** | 2/5 | HARM | **4/5** | 1/5 | HARM | + +(Full 24-row table in `SCRATCH/w3b_indexes/logs/fixreport_full.txt`.) + +| subset | n | OFF | ON | 2026-08-09 OFF | 2026-08-09 ON | +|---|---|---|---|---|---| +| all rows | 24 | 10/24 | 10/24 | 11/24 | 10/24 | +| fired rows | 2 | **2/2** | **0/2** | 1/1 | 0/1 | + +Whole-set is a wash both dates, as expected (22 of 24 rows are identical work in both +arms). The `acq` whole-set barely moved across dates (11/24 โ†’ 10/24 OFF), consistent +with ยง2.1: this corpus's prompts largely fit even before the fix. **Every fire under +the true product default, on both dates, is a mechanical regression** โ€” now 2 for 2. + +--- + +## 6. The measurement caveat that qualifies ยง4 and ยง5 โ€” and does not rescue 4.1 + +The decision file's ยง7 warned the judge is "strict and, in this framing, noisy." On +these arms it is worse than noisy: **on the fired rows it returns verdicts its own +stated reasons contradict.** Every ON-arm "HARM" above was hand-checked against gold. + +`acq_q04`, cell 1, ON arm, judged **0/5**. Verbatim answer: + +> "the single largest client is **MegaCorp**, which accounts for **28%** of +> StartupXYZ's revenue" + +against gold *"The largest customer, MegaCorp, accounts for 28% of revenue."* The gold +fact is present verbatim. Two of the five judge reasons are flatly false โ€” *"the +EVIDENCE explicitly states that the single largest client is TechCorp Industries"* and +*"contains no information about MegaCorp's name or its 28% revenue share."* A third +cites the verifier's appended `[Confidence: 90%]` suffix as grounds for rejection. + +`acq_q12`, cell 2, ON arm, judged **2/5**, answer contains *"47 employees โ€ฆ Engineering +with 32 โ€ฆ Sales with 8 โ€ฆ Operations with 7"* โ€” the complete gold fact โ€” while two of +its three sampled reasons say the answer *"accurately reflects all specific workforce +numbers and functional splits."* The verdict contradicts the reason. + +The verifier's `[Confidence: N%] [Warning: โ€ฆ Groundedness: False]` suffix appears on +7/9 answers in **both** cell-1 arms, so it is a noise source, not an arm-specific bias. + +**Manual adjudication of every fired row** (is the gold fact present in the answer?): + +| cell | qid | OFF | ON | +|---|---|---|---| +| 1 | `acq_q04` | yes | yes | +| 1 | `acq_q07` | yes | yes (in a 35 290-char runaway answer) | +| 1 | `acq_q09` | yes | yes, but prefaced *"there is no information regarding borrowing"* | +| 1 | `acq_q12` | no (abstains) | no (abstains) | +| 1 | `docs_d03` | yes | yes | +| 1 | `docs_d05` | yes | yes | +| 1 | `docs_d13` | no (partial) | no | +| 2 | `acq_q04` | yes | yes | +| 2 | `acq_q12` | yes | yes | + +**Manual: 7/9 โ†’ 7/9 across all fired rows in both cells โ€” an exact wash.** +**Mechanical judge: 7/9 โ†’ 2/9.** + +The two instruments disagree on whether escalation *harms*. They agree completely on +the only question the HOLD asked: **there is no lift.** No fired row in either cell +gains a gold fact it did not already have without escalation. + +--- + +## 7. Proposed verdict + +### **4.1 full-document escalation โ€” do not adopt. Convert the HOLD to a REJECT as a shipped default; keep the flag and the code.** + +**The one line it rests on:** the 0/7 โ†’ 2/7 lift was produced by front-truncation, not +by document reassembly โ€” with prompts that actually fit, the escalation-off baseline +on the identical fire subset goes from 0/9 to **7/9** and escalation adds nothing on +top of it (**5/7 โ†’ 2/7** mechanically, **5/7 โ†’ 5/7** by hand), while both fires under +the true product default remain regressions. + +Supporting points: + +* The HOLD was explicitly conditional on the lift surviving. It did not survive; it + inverted. The condition resolves to "do not adopt." +* The mechanism ยง6.2 hypothesised is now confirmed rather than inferred: the escalated + block survived truncation because it was appended at the tail while top-ranked + chunks were deleted from the front. Remove the truncation and the benefit vanishes. +* The real fix for this corpus was the context-window bug, exactly as ยง8 predicted + ("any context-window guard โ€ฆ would change the answers on this corpus more than + either Phase-4 feature does"). It moved the deciding cell by **7 rows**; escalation + moved it by 0 (manual) to โˆ’3 (mechanical). +* Evidence for active *harm* is weak and should not be cited: all four mechanical + HARM rows are judge artifacts on inspection (ยง6). "No benefit" is the defensible + claim; "harmful" is not. +* Keep the code behind the flag. It is unfalsified for the case it was designed for + (a prompt that fits *and* a document whose ordering matters); this corpus at + `chunk_size=512` with a 32k window simply never presents that case. + +### Confidence and limits + +* n = 7 fired rows in the deciding cell, 2 in the product-default cell. Small, as + before. But the decisive number is not the fired-row delta โ€” it is the **7-row move + in the baseline**, which is far outside the ~1-row noise floor ยง7 established. +* The judge remains the weakest instrument here. The verdict is stated so that it + holds under *both* the mechanical and the manual reading. +* No wall-clock number is used as evidence. +* `atlas7` / `hr` still excluded (1 and 2 chunks). 4.2 and 4.3 untouched. + +--- + +## 8. Anomalies and new backlog + +1. **`acq_q07`, cell 1, escalation ON, produced a 35 290-character answer** (vs 1 400 + in the OFF arm) that degenerates into verbatim regurgitation of + `design_rationale.md` text about MiniCheck โ€” unrelated to the HSR question it + answered correctly in its first paragraph. A whole-document block in context can + send the 9b model into transcription. New failure mode, only visible now that + prompts are not truncated. +2. **The judge returns verdicts contradicted by its own reasons** on multi-clause gold + facts (ยง6), and is measurably perturbed by the verifier's appended + `[Confidence: โ€ฆ] [Warning: โ€ฆ Groundedness: False]` suffix โ€” one reason cites the + 90% confidence figure as grounds for rejecting the answer. The suffix is part of + the answer string the judge scores. Either strip it before judging or stop + appending it to the user-visible answer. +3. **The 2026-08-09 `acqdocs` numbers in `eval/decisions/phase4-answer-quality.md` ยง4 + should be treated as void**, not merely confounded: that baseline was 0/9 because + it was reading truncated context, and it is 7/9 on identical inputs today. ยง3 + (`acq`) is largely unaffected (ยง2.1). +4. The `think:false` synthesis-path bug (ยง8 of the decision file) is still unfixed and + still requires the harness monkeypatch. + +--- + +## 9. Files produced (all under SCRATCH, nothing in the repo) + +``` +w3b_indexes/results/ad_fire_esc_on_fix.jsonl ad_fire_esc_off_fix.jsonl +w3b_indexes/results/acq_esc_on_fix.jsonl acq_esc_off_fix.jsonl +w3b_indexes/results/judge5_*_fix.jsonl (k=5 verdicts, 4 files) +w3b_indexes/results/ctx_*_fix.jsonl (249 per-call num_ctx / prompt_eval_count records) +w3b_indexes/logs/run_*_fix.log, *_fix.stdout.log +w3b_indexes/logs/fixreport_full.txt (full 24-row cell-2 table) +w3b_ctxprobe.py w3b_run_fix.py w3b_fixreport.py chain_fix1.sh chain_fix1b.sh +``` + +Original 2026-08-09 `*.jsonl` arm outputs were read but never modified. + +### Execution note + +The first background chain was killed by the harness partway through +`acq_esc_off_fix` (20 of 24 rows written) and a `setsid`-based relaunch failed +(`nohup: setsid: No such file or directory` โ€” not present on macOS). It was resumed +with plain `nohup`; `w3b_run.py` skips ids already present in its `--out` file, so +rows 21โ€“24 of that arm ran in a second process. Consequence: the per-model `num_ctx` +ratchet restarted for those 4 rows. Since sizing is per-prompt and the ratchet only +ever *grows* the window, no prompt was under-sized โ€” 0 warnings across all 102 calls +of that arm. No query ran twice; no row was overwritten. + +--- + +## 10. Gate validation (2026-08-12) + +Every deciding number above was independently recomputed from the raw +`*_fix.jsonl` files by the gate, trusting nothing in this report's prose: + +* Cell 1 per-row `k/5` table and majorities: reproduced exactly (7/9 OFF, 4/9 ON; + fired subset 5/7 โ†’ 2/7). +* Cell 2 majorities and fired rows: reproduced exactly (10/24 both arms; + `acq_q04` 4/5 โ†’ 0/5, `acq_q12` 4/5 โ†’ 2/5). +* Fire sets: extracted from the `document_escalation` payloads directly โ€” + cell 1 fired 7 (`+acq_q07`, `โˆ’docs_d20` vs 2026-08-09), cell 2 fired + `acq_q04` + `acq_q12`, OFF arms fired zero. +* Truncation telemetry: 249 calls, max `prompt_eval_count` 27 516, zero rows at + `num_ctx โˆ’ 16`. Confirmed from `ctx_*_fix.jsonl`, not the report. +* Judge-artifact claims spot-checked against raw rows: `acq_q04` (cell 1, ON) is + 0/5 despite containing gold verbatim ("MegaCorp, which accounts for 28%"); + `acq_q12` (cell 2, ON) verdicts contradict their own sampled reasons. +* `acq_q07` runaway answer: 35 290 chars ON vs 1 400 OFF, confirmed from the raw + arm file. + +Verdict accepted as proposed: **4.1 REJECTED as a shipped default** (was HOLD); +flag and code kept. Doc tables updated in `research_roadmap.md` and +`design_rationale.md` ยง13a in the same commit. diff --git a/eval/decisions/phase4-escalation-tokens.md b/eval/decisions/phase4-escalation-tokens.md new file mode 100644 index 00000000..d7108e5c --- /dev/null +++ b/eval/decisions/phase4-escalation-tokens.md @@ -0,0 +1,416 @@ +# Phase 4.1 + 4.5 โ€” full-document escalation and per-query token tracking โ€” implemented 2026-08-09 + +**Status of each item, stated up front:** + +| Item | Ships as | Default | +|---|---|---| +| 4.1 full-document escalation | flag-gated code path, **unmeasured** | **OFF** | +| 4.5 per-query token tracking | always-on observability | **ON** | + +4.1 has **not** been benchmarked against the gold set. It is off, and it should +stay off until someone runs the A/B described in ยง6. Nothing in this document +claims it improves answers; it claims only that it fires when it is supposed to +and produces the document it says it produces. + +No file in `Documentation/` was edited this wave. Proposed documentation diffs +are in ยง7, to be applied at the adoption gate. + +--- + +## 1. What shipped + +### 4.1 โ€” full-document escalation (off by default) + +New files: + +* **`rag_system/retrieval/document_fetch.py`** โ€” reassembles one document from + its indexed chunks. Filters the LanceDB text table by `document_id`, orders by + `chunk_index`, prefers `metadata.metadata.original_text` over the top-level + `text` column (which carries the contextual-enrichment `Context: โ€ฆ` preamble + when enrichment is on), and truncates to a token budget. Returns `None` + โ€” never raises โ€” when the table cannot be opened, the document has no rows, or + no row carries a usable `chunk_index`. **Order is the point**: a document whose + chunks cannot be ordered is not escalated at all rather than escalated + scrambled (DOS-RAG). +* **`rag_system/agent/escalation.py`** โ€” `EscalatingRetrievalPipeline`, the + trigger and the wiring. + +**Token counting is `len(text) // 4`**, stated as such in the module docstring +and surfaced everywhere as `approx_tokens`. No tokenizer is loaded. The budget is +a context-window guard, not an accounting figure, and a 4-chars-per-token +estimate runs slightly conservative on English prose. + +**Trigger.** After candidate selection completes โ€” that is, *after* the +evidence-sufficiency retry (ยง5 of `design_rationale.md`) has had its one attempt +โ€” the same signal the retry uses is read again: the reranker's calibrated +probability when there is one, else the dense contrast score +`(cos_top โˆ’ cos_background) / (1 โˆ’ cos_background)`. If it is still below +threshold, the top-ranked chunk's whole document is reassembled and appended to +the synthesis context as: + +``` +โ€“โ€“โ€“โ€“โ€“ FULL DOCUMENT (escalated): <name> โ€“โ€“โ€“โ€“โ€“ +<document text, in chunk order> +โ€“โ€“โ€“โ€“โ€“ END FULL DOCUMENT โ€“โ€“โ€“โ€“โ€“ +``` + +The threshold defaults to the retry's own `min_top_score` (0.12) โ€” escalation is +what happens when the retry already ran and the evidence is *still* weak, so the +two are judged against the same bar unless `min_evidence` overrides it. Where the +retry has no signal (`fts_only`, legacy unnormalized tables), escalation has no +signal either and does not fire โ€” same rule, deliberately. + +**Bounds.** One document per user query (`max_documents`, enforced by a locked +per-request budget so a decomposed query's parallel sub-queries cannot each +escalate), one token budget, no loop, no LLM call of its own. + +**Citations are untouched.** `source_documents` is exactly what it would have +been without escalation. The escalated block is reading material appended to the +synthesis prompt, not a source. + +**Event.** `document_escalation` is emitted through the existing +`event_callback`, so it lands in the SSE stream next to `retrieval_retry`. +Payload: `document_id`, `document_name`, `chunks_used`, `chunks_total`, +`approx_tokens`, `truncated`, `signal`, `score`, `threshold`, `token_budget` โ€” +never the document text. `/chat` (non-streaming, no events) gets the same +payloads as a `document_escalation` list on the result. + +### 4.5 โ€” per-query token tracking (on by default) + +* **`rag_system/utils/ollama_client.py`** โ€” `generate_completion`, + `generate_completion_async` and `stream_completion` now hand Ollama's + `prompt_eval_count` / `eval_count` to a `TokenUsageTracker` bound for the + duration of one user query. `stream_completion` also takes an optional + `stats` dict out-parameter, since a generator cannot return the final object. +* **`rag_system/agent/loop.py`** โ€” one tracker per `Agent.run()`, with stage + labels around triage, decomposition, synthesis and verification. The result + payload (and therefore the SSE `complete` event and the `/chat` response body) + carries: + +```json +"token_usage": { + "by_stage": {"synthesis": {"prompt_tokens": 634, "output_tokens": 430, "calls": 1}}, + "total": {"prompt_tokens": 634, "output_tokens": 430, "calls": 1, "total_tokens": 1064} +} +``` + +* **`rag_system/utils/watsonx_client.py`** โ€” returns `prompt_eval_count: 0` / + `eval_count: 0` and records the call, so a watsonx run reports an honest + "N calls, 0 tokens counted" rather than looking like a cache hit. +* **`rag_system/api_server.py`** โ€” **no change was needed.** It already returns + the agent's result dict verbatim from `/chat` and passes it as the `complete` + event's data, so `token_usage` flows through both endpoints for free. This was + verified over HTTP (ยง4). + +--- + +## 2. Honest limitations of the token accounting + +* **The retry's reformulation call is counted under `synthesis`.** It is an + enrichment-model call made *inside* `RetrievalPipeline.run()`, which the agent + wraps as one "synthesis" stage. Splitting it would require editing + `retrieval_pipeline.py`, which this wave does not own. +* **There is no `escalation` stage bucket**, because escalation makes no LLM call + of its own by design โ€” it enlarges the synthesis prompt. Its cost shows up as + a larger `synthesis.prompt_tokens`, and the size of the appended block is in + the `document_escalation` event's `approx_tokens`. A stage that never fires + would have been a lie; an empty bucket is not emitted. +* **`by_stage` omits stages that made no call.** An absent key means "no LLM call + in that stage", not "zero tokens". +* **Only Ollama reports real counts.** watsonx reports zeros (see above). +* **Embedding calls are not counted.** Ollama's embedding endpoint is not routed + through these three methods, and the shipped embedder is in-process + (harrier-oss-v1) so there is no token count to read. + +--- + +## 3. The ownership question โ€” `retrieval_pipeline.py` was NOT touched + +**This wave made zero edits to `rag_system/pipelines/retrieval_pipeline.py`.** +(The file does carry uncommitted changes in the working tree โ€” they belong to +another workstream, not to 4.1/4.5. Nothing below touched it.) + +This is worth flagging because the brief anticipated a small additive change +there and it turned out to be avoidable. The problem: escalation has to happen +*between* candidate selection and synthesis, and both live inside +`RetrievalPipeline.run()`, which returns only `{"answer", "source_documents"}`. +Exposing the contrast signal on `retrieve_candidates()`'s return value would not +have been enough โ€” `run()` does not pass it out, and by the time the agent sees +the result the answer has already been generated. Escalating from the agent loop +after `run()` returns would have meant a **second** synthesis pass: two +generations, two token streams into the UI. + +Instead, `Agent` now constructs `EscalatingRetrievalPipeline` โ€” a subclass that +overrides the two methods `run()` already calls in sequence: + +* `retrieve_candidates()` โ†’ `super()`, then read the final post-retry evidence + score and remember the top-ranked chunk's `document_id`; +* `_synthesize_final_answer()` โ†’ append the document block to `facts`, then + `super()`. + +Result: single generation pass, zero edits to the other workstream's file. The +handoff between the two hooks is a `threading.local`, so the parallel sub-query +fan-out cannot cross-wire one sub-query's document into another's synthesis. + +**What the gate must know:** this couples escalation to two method names in +`retrieval_pipeline.py` โ€” the public `retrieve_candidates()` and the private +`_synthesize_final_answer(query, facts, *, event_callback=None)`. If either is +renamed or its signature changes, escalation stops firing. The constructor logs +a warning when `retrieve_candidates` is absent, and `_plan_escalation` is wrapped +so a failure there can never break retrieval โ€” but a rename would be a silent +loss of the feature, not a crash. If the pipeline is being restructured anyway, +promoting these to a documented seam (e.g. an `extra_context` hook before +synthesis) is the cleaner long-term shape. + +--- + +## 4. Verification + +All of this ran on this machine on 2026-08-09. Nothing below is estimated. + +### 4a. Static checks + +* `python -m py_compile` on every touched Python file: clean. +* `npx tsc --noEmit` (`src/` was touched): exit 0, no output. + +### 4b. `eval/smoke_e2e.py` + +``` +$ .venv/bin/python eval/smoke_e2e.py +... + wall clock 261.2s + +======================================================================== +25/25 assertions passed +======================================================================== +``` + +Unchanged from the pre-change baseline, as expected: escalation is off, and +token tracking adds a key nobody asserts on. + +### 4c. Scratch test โ€” escalation ON against a throwaway Atlas-7 index + +Not added to `eval/` (owned by another workstream this wave). The script built a +throwaway LanceDB index from `eval/corpora/atlas7_service_manual.pdf` +(docling chunker, 100-token chunks โ†’ 5 chunks), then drove the agent and a live +RAG API on port 8077. Verbatim output: + +``` +--- index has 5 chunks for atlas7_service_manual.pdf --- +PASS 4.1b fetch returns a document +PASS 4.1b all chunks reassembled | used=5 total=5 rows=5 +PASS 4.1b chunk order is ascending | indices=[0, 1, 2, 3, 4] +PASS 4.1b not truncated at a large budget +PASS 4.1b document text follows chunk_index order +PASS 4.1b truncates to the token budget | approx_tokens=300 budget=300 truncated=True +PASS 4.1b truncated text is a prefix of the full text +PASS 4.1a escalation is OFF by default | events=['analyze', 'retrieval_done', 'retrieval_retry', 'retrieval_started', 'token'] +PASS 4.1a no escalation payload in the default result +PASS 4.1a escalation fires on a weak query | events=['analyze', 'document_escalation', 'retrieval_done', 'retrieval_retry', 'retrieval_started', 'token'] +PASS 4.1a event carries name/score/threshold | {"document_id": "atlas7_service_manual.pdf", "document_name": "atlas7_service_manual.pdf", "chunks_used": 5, "chunks_total": 5, "approx_tokens": 330, "truncated": false, "signal": "dense_contrast", "score": 0.0431, "threshold": 0.12, "token_budget": 6000} +PASS 4.1a score is below the threshold | 0.0431 < 0.12 +PASS 4.1a within the token budget | approx_tokens=330 +PASS 4.1a exactly one document escalated | count=1 +PASS 4.1a citations survive escalation | sources=5 +PASS 4.1a strong query does not escalate | events=['analyze', 'retrieval_done', 'retrieval_started', 'token'] +PASS 4.5 token_usage present in the agent result | {"by_stage": {"synthesis": {"prompt_tokens": 1084, "output_tokens": 3551, "calls": 2}}, "total": {"prompt_tokens": 1084, "output_tokens": 3551, "calls": 2, "total_tokens": 4635}} +PASS 4.5 has by_stage + total +PASS 4.5 non-zero totals | {"prompt_tokens": 1084, "output_tokens": 3551, "calls": 2, "total_tokens": 4635} +PASS 4.5 synthesis stage attributed | stages=['synthesis'] + token_usage = {"by_stage": {"synthesis": {"prompt_tokens": 1084, "output_tokens": 3551, "calls": 2}}, "total": {"prompt_tokens": 1084, "output_tokens": 3551, "calls": 2, "total_tokens": 4635}} +PASS 4.5c RAG API started +PASS 4.5c /chat returned 200 +PASS 4.5c token_usage in the /chat response body +PASS 4.5c totals are non-zero | {"by_stage": {"synthesis": {"prompt_tokens": 1192, "output_tokens": 861, "calls": 1}}, "total": {"prompt_tokens": 1192, "output_tokens": 861, "calls": 1, "total_tokens": 2053}} + /chat token_usage = {"by_stage": {"synthesis": {"prompt_tokens": 1192, "output_tokens": 861, "calls": 1}}, "total": {"prompt_tokens": 1192, "output_tokens": 861, "calls": 1, "total_tokens": 2053}} + +ALL PASSED +``` + +What each group establishes: + +* **(a) it fires on a weak query.** `"summarize the overall approach and its + implications"` against a coffee-machine service manual scored + `dense_contrast = 0.0431` against a `0.12` threshold *after* the retry had + already run and failed to improve it (`retrieval_retry` is in the event list + on both the off and on runs). With the flag off no `document_escalation` + event and no result key; with it on, exactly one. A well-matched query + ("What pressure does the brew boiler operate at during extraction?") scored + above threshold and did **not** escalate โ€” the trigger discriminates, it does + not just fire on everything. +* **(b) in-order and inside the budget.** Reassembly used all 5 chunks with + `chunk_indices == [0,1,2,3,4]`, and each chunk's text was located in the + assembled string strictly after the previous chunk's โ€” so the output is + genuinely in `chunk_index` order, not merely composed of the right pieces. At + a deliberately tiny 300-token budget it truncated to exactly `approx_tokens=300`, + set `truncated=True`, and the truncated text is a prefix of the untruncated + text. +* **(c) `token_usage` in a real `/chat` response.** Over HTTP against a RAG API + started as a subprocess: `1192` prompt / `861` output tokens. + +Two things this run also confirms about ยง2's caveats, visible in the numbers: +`calls: 2` on the escalated in-process run is the retry's reformulation call +*plus* the synthesis call, both billed to `synthesis`; and there is no +`escalation` bucket, because escalation made no LLM call of its own. + +**What this does NOT establish:** whether the escalated answer is *better*. No +answer-quality comparison was run. See ยง6. + +--- + +## 5. Proposed config keys (for `rag_system/main.py`, at the adoption gate) + +`main.py` was not touched. Every key below is read through `config.get()` with +the default baked into `DEFAULT_DOCUMENT_ESCALATION` in +`rag_system/agent/escalation.py`, so the code behaves identically whether or not +the profiles declare them. Adding them makes the flag discoverable and +togglable per profile: + +```python +# in PIPELINE_CONFIGS["default"]["retrieval"], next to "retry": + + # Full-document escalation (roadmap 4.1). OFF until benchmarked. + # When the evidence-sufficiency retry above has already run and the + # evidence is STILL below threshold, reassemble the top-ranked + # chunk's whole document in chunk_index order and append it to the + # synthesis context, capped at token_budget. One document, no loop. + # min_evidence defaults to retry.min_top_score when omitted. + "document_escalation": { + "enabled": False, + "max_documents": 1, + "token_budget": 6000 + } +``` + +```python +# in PIPELINE_CONFIGS["fast"]["retrieval"], next to "retry": + + # Off in `fast` for the same reason the retry is: this profile + # exists to avoid extra work, and escalation only inflates the + # synthesis prompt. + "document_escalation": {"enabled": False} +``` + +Optional fourth key, not declared above because the fallback is the better +default: `"min_evidence": <float>` overrides the trigger threshold +independently of `retry.min_top_score`. + +No environment variable was added โ€” the brief did not ask for one and a flag +that is off pending measurement does not need a second way to turn it on. + +--- + +## 6. What must be measured before 4.1 is switched on + +The flag exists so this can be answered with numbers rather than intuition: + +1. **Answer quality on weak-evidence queries.** Run the gold set with + `document_escalation.enabled` false/true and judge only the subset where the + trigger actually fires (on most queries the run is byte-identical, so a + whole-set average would drown the effect). The relevant comparison is + end-to-end answer correctness, not retrieval metrics โ€” escalation changes no + retrieval output, only the synthesis prompt. +2. **The cost.** `token_usage.by_stage.synthesis.prompt_tokens` with and without, + plus wall-clock. A 6000-token budget is a large prompt increase on a local + model, and long-context degradation ("lost in the middle") is a real risk on + a small generation model. +3. **The budget itself.** 6000 is a guess. It should be tuned against the + generation model's context window and the corpus's document lengths. +4. **Whether the threshold should differ from the retry's.** Inheriting + `min_top_score` is a defensible default, not a measured one. + +Until (1) shows a win on the fire-subset, the honest status is: implemented, +bounded, unmeasured, off. + +--- + +## 7. Proposed documentation diffs (NOT applied โ€” apply at adoption) + +Docs describe shipped behaviour, and 4.1 is off, so **the only doc change that +should land before the escalation A/B is the 4.5 one.** + +### 7a. `Documentation/design_rationale.md` โ€” new section, ship now (4.5 is on) + +Insert after ยง5 (evidence-sufficiency retry): + +```diff ++## 5a. Per-query token accounting ++ ++**What ships.** Every Ollama completion the agent makes โ€” streaming or not โ€” ++reports `prompt_eval_count` and `eval_count` on its final object. Those are ++aggregated per user query, bucketed by pipeline stage (`triage`, ++`decomposition`, `synthesis`, `verification`), and returned as `token_usage` on ++the `/chat` response body and in the SSE `complete` event. On by default: it ++costs one dict update per LLM call and adds no request. ++ ++The aggregation point is a `ContextVar` in `rag_system/utils/ollama_client.py` ++rather than an argument threaded through every call site, because one ++`OllamaClient` is shared by the agent, the retrieval pipeline, the verifier and ++the decomposer. `await` and `asyncio.to_thread` propagate it; the agent's ++parallel sub-query `ThreadPoolExecutor` copies it explicitly. ++ ++Two honest gaps: the retry's reformulation call is billed to `synthesis`, ++because it happens inside `RetrievalPipeline.run()` which the agent labels as ++one stage; and watsonx reports zeros, because the SDK path in use surfaces no ++per-call counts. +``` + +### 7b. `Documentation/api_reference.md` โ€” ship now + +Add `token_usage` to the documented `/chat` response body and to the +`complete` SSE event's data, with the shape shown in ยง1 above, and the note that +an absent stage key means "no LLM call in that stage". + +### 7c. `Documentation/design_rationale.md` โ€” 4.1, hold until measured + +When (and only when) the ยง6 A/B shows a win: + +```diff ++## 5b. Full-document escalation ++ ++**What ships.** `retrieval.document_escalation` โ€” <default to be set by the ++A/B>. When the evidence-sufficiency retry (ยง5) has run and the signal is still ++below threshold, the top-ranked chunk's document is reassembled from the index ++in `chunk_index` order and appended to the synthesis context as one delimited ++block, capped at `token_budget` and at one document per query. Chunk citations ++are unchanged. Surfaces as a `document_escalation` SSE event. ++ ++**Why in-order.** DOS-RAG: a document handed to the model in its original order ++beats the same text ranked by similarity. A document whose chunks carry no ++usable `chunk_index` is therefore not escalated at all. ++ ++**Why bounded.** PEA-CAE's escalate-don't-pre-decide, without the loop: ++search volume correlates only weakly with answer quality, so the escalation is ++one document, once, with no agency over what to read next. +``` + +### 7d. `Documentation/research_roadmap.md` โ€” at adoption + +Mark 4.5 done and 4.1 as implemented-but-gated in the Phase 4 table. + +--- + +## 8. Backlog created by this wave + +* **UI does not display token counts.** `token_usage` reaches the browser on the + `complete` event and is persisted into the turn's steps snapshot via the + existing `saveStreamedTurn` path (`src/components/ui/session-chat.tsx`, final + step's `details.token_usage`). Rendering it as a compact + "ยท 1.2k in / 340 out" line needs a change in + `src/components/ui/conversation-page.tsx`, which renders the cascade โ€” out of + scope for a "minimal `src/`" wave. +* **The non-streaming gateway path drops `token_usage`.** The browser streams + straight from the RAG API (`RAG_API_BASE_URL/chat/stream`), so the SSE + `complete` event carries it. But `backend/server.py`'s `_query_rag_api` + extracts only `answer` and `source_documents` from the RAG API's `/chat` + response, so the non-streaming fallback loses it. One line in `backend/` + (not owned this wave) would fix it. +* **Pre-existing off-by-one in the step cascade.** `session-chat.tsx` addresses + steps positionally (`steps[5]`, `steps[6]`, `steps[7]`) in the + `sub_query_result`, `sub_query_token`, `final_answer` and `token` handlers. + Those indices were correct before the `retry` step was inserted at index 3 and + are now one short โ€” `sub_query_result` writes sub-answers into the + "Expanding context window" step. **Not introduced by this wave and not fixed by + it**; the new `document_escalation` handler deliberately uses `findIndex` and + adds no array entry, precisely so it does not shift these further. Worth a + dedicated fix that converts every positional access to `findIndex`. diff --git a/eval/decisions/phase4-eval-final-metric.md b/eval/decisions/phase4-eval-final-metric.md new file mode 100644 index 00000000..a39ef43d --- /dev/null +++ b/eval/decisions/phase4-eval-final-metric.md @@ -0,0 +1,339 @@ +# Phase 4 item 4.2 โ€” making the cross-reference hop measurable, and the first A/B + +Date: 2026-08-09 +Status: **measurement infrastructure landed in `eval/run_eval.py`; the 4.2 A/B was +run and is a NEGATIVE result โ€” the hop fires zero times on the corpus built to +exercise it.** No adopt/reject call is made here; that is the gate's. +Scope of this wave: `eval/run_eval.py`, `eval/BASELINE.md`, this file. No +`Documentation/` diffs, no `rag_system/` edits. + +--- + +## 1. The problem this wave was given + +`eval/run_eval.py` called `pipeline.retrieve_candidates(...)` and scored +`out["first_stage"]` (plus `ndcg10_reranked` off the reranked list). The +cross-reference hop +(`rag_system/pipelines/retrieval_pipeline.py::_crossref_hop`) appends its hopped +chunks to `result["documents"]` and **deliberately never mutates +`first_stage`** โ€” the previous decision file +([`phase4-crossref-prefilter.md`](phase4-crossref-prefilter.md) ยง5) says so in +as many words: *"The harness will therefore report a flat line for 4.2 no matter +how well it works."* + +## 2. What changed in `eval/run_eval.py` + +### 2.1 The final candidate list is now scored + +Every query is scored twice, against two lists: + +| metric family | source | meaning | +|---|---|---| +| `recall` / `ndcg10_first_stage` | `out["first_stage"]` | the retriever's own ordering (unchanged; every historical number in `BASELINE.md` still means the same thing) | +| `recall_final` / `ndcg10_final` | `out["documents"]` | **post-rerank AND post-hop** โ€” the list the answer stage would actually see | + +Summary keys: `recall@{5,10,20}_final`, `ndcg@10_final`, present on the whole-corpus +summary, on the `crossref` / `crossref_control` slices, and in `by_dimension` +(`recall@10_final`, `ndcg@10_final`, `queries_with_crossref_hop`). + +`ndcg10_reranked` / `recall_reranked` were **kept meaning post-rerank, pre-hop**. +The hop only ever appends, so dropping the `via_crossref`-tagged rows +reconstructs the pre-hop list exactly; every earlier decision file's +`ndcg@10_reranked` stays comparable. + +### 2.2 The invariant, checked rather than argued + +With reranking off and the hop off, `documents` **is** `first_stage` (the same +list object). Every run now records `final_equals_first_stage` per query +(chunk-id sequence equality) and prints a run-level verdict, also written to the +results JSON as `final_vs_first_stage_invariant`. If the final metrics ever drift +from the first-stage metrics on a run where nothing may reorder or append, the +metric is measuring its own bug. + +### 2.3 Hop instrumentation โ€” precision, not just rank movement + +Per query, when anything was hopped: + +* `crossref_chunks_in_final` โ€” how many `via_crossref` chunks are in `documents` +* `crossref_documents` โ€” which documents they came from +* `crossref_hit_expected_source` โ€” did a hop land in a document named in the gold + row's `expected_sources`? (document-level precision) +* `crossref_chunk_relevant` โ€” does a hopped chunk actually contain the gold + `expected` text? (text-level precision) +* `first_relevant_rank_final`, and the raw `crossref_hop` record from the pipeline + +Aggregated per corpus and per slice as `crossref_hop: {queries_with_hop, +fire_rate, chunks_added_total, chunks_added_mean_when_fired, hit_expected_source, +hopped_chunk_relevant}`. + +### 2.4 CLI toggles for the Phase-4 flags + +Following the existing `--retry {profile,on,off}` pattern, via a new +`apply_phase4_settings()` that writes `retrieval.crossref_hop` / +`retrieval.overview_prefilter` the same way `apply_retry_setting` writes +`retrieval.retry`: + +``` +--crossref-hop {profile,on,off} +--overview-prefilter {profile,off,boost,restrict} +``` + +`profile` = whatever `main.py` says (both are OFF there today). Both states are +echoed in the run header and recorded in the results JSON (`run.crossref_hop`, +`run.overview_prefilter` and their config blocks). + +--- + +## 3. Verification + +Determinism protocol for everything below: `--retry off`, no empty-string env +vars, no `Documentation/` edits between runs. All indexes were reused from cache +(`acq` 13 chunks, `acq+docs` 373 chunks, `mixed` 363 chunks), so every arm below +is a code-to-code comparison on identical bytes. + +### 3.1 Regression โ€” `mixed`, all Phase-4 flags off + +``` +.venv/bin/python eval/run_eval.py --corpus mixed --retry off \ + --crossref-hop off --overview-prefilter off \ + --json-out eval/results/phase4_finalmetric_regression_mixed.json +``` + +``` +corpus n chunks R@5 R@10 R@20 nDCG@10 nDCG@10 | R@5 R@10 R@20 nDCG@10 hop q 1st ms + (1st) (rerank) | (fin) (fin) (fin) (final) +------------------------------------------------------------------------------------------------------------------------- +mixed 72 363 0.944 0.972 1.000 0.887 n/a | 0.944 0.972 1.000 0.887 0 124 + +invariant โœ… final == first_stage on all 72 queries (rerank OFF, crossref hop OFF) โ€” chunk-id order and both metrics +``` + +First stage is **identical** to the tracked baseline (mixed, 363 chunks: 0.944 / +0.972 / 1.000, nDCG@10 first-stage 0.887 โ€” `phase4-crossref-prefilter.md` ยง2, the +retry-off arm). Final metrics equal first-stage metrics on all 72 queries. + +### 3.2 The 4.2 A/B โ€” `acq`, hop off vs hop on + +``` +.venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop off \ + --overview-prefilter off --json-out eval/results/phase4_42_acq_hop_off.json +.venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop on \ + --overview-prefilter off --json-out eval/results/phase4_42_acq_hop_on.json +``` + +| arm | slice | n | R@5 | R@10 | R@20 | nDCG@10 (1st) | R@5 (fin) | R@10 (fin) | R@20 (fin) | **nDCG@10 (final)** | queries that hopped | +|---|---|---|---|---|---|---|---|---|---|---|---| +| hop **off** | all | 24 | 0.958 | 1.000 | 1.000 | 0.8101 | 0.958 | 1.000 | 1.000 | **0.8101** | 0 | +| hop **off** | `requires_crossref=true` | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | 0 | +| hop **off** | control (`=false`) | 13 | 0.923 | 1.000 | 1.000 | 0.8628 | 0.923 | 1.000 | 1.000 | **0.8628** | 0 | +| hop **on** | all | 24 | 0.958 | 1.000 | 1.000 | 0.8101 | 0.958 | 1.000 | 1.000 | **0.8101** | **0** | +| hop **on** | `requires_crossref=true` | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | **0** | +| hop **on** | control (`=false`) | 13 | 0.923 | 1.000 | 1.000 | 0.8628 | 0.923 | 1.000 | 1.000 | **0.8628** | **0** | + +First stage is bit-identical across arms, as designed. **Final metrics are also +identical, because the hop fired on 0 of 24 queries.** Two independent reasons, +both verified: + +1. **`acq` is 13 chunks.** At `k = 20` every query retrieves all 13, so all ten + documents are already `represented` and the hop's "not already a candidate" + guard suppresses every target. Confirmed: `candidates` = 13 and + `final_candidates` = 13 on all 24 queries. +2. **More fundamental โ€” none of `acq`'s cross-references resolve.** Read straight + out of the built index: + + ``` + acq index: 13 chunks, 11 chunks carrying crossrefs, 34 refs + total refs 34 resolved 0 + top unresolved on acq: [('exhibit','exhibit c') x6, ('exhibit','schedule 3') x5, + ('exhibit','schedule 1') x5, ('exhibit','exhibit a') x4, ('exhibit','exhibit b') x4, + ('exhibit','schedule 2') x4, ('section','section 2.2') x2, ('section','section 1.5'), + ('section','section 4.1'), ('section','section 1.1'), ('section','section 4')] + ``` + + Reason 2 is why lowering `k` does not rescue it. Both `--k 5` and `--k 3` were + run on both arms (`phase4_42_acq_k{5,3}_hop_{off,on}.json`) โ€” at `k = 3` only + 3 of 13 chunks are candidates, so most documents are unrepresented, and the hop + *still* fired 0 times. Every number is identical across arms: + + | arm | slice | n | R@5 | nDCG@10 (1st) | R@5 (fin) | nDCG@10 (final) | hops | + |---|---|---|---|---|---|---|---| + | k=5 hop off | all | 24 | 0.917 | 0.7875 | 0.917 | 0.7875 | 0 | + | k=5 hop on | all | 24 | 0.917 | 0.7875 | 0.917 | 0.7875 | 0 | + | k=5 hop off | xref | 11 | 1.000 | 0.7358 | 1.000 | 0.7358 | 0 | + | k=5 hop on | xref | 11 | 1.000 | 0.7358 | 1.000 | 0.7358 | 0 | + | k=3 hop off | all | 24 | 0.792 | 0.8036 | 0.792 | 0.8036 | 0 | + | k=3 hop on | all | 24 | 0.792 | 0.8036 | 0.792 | 0.8036 | 0 | + | k=3 hop off | xref | 11 | 0.909 | 0.7868 | 0.909 | 0.7868 | 0 | + | k=3 hop on | xref | 11 | 0.909 | 0.7868 | 0.909 | 0.7868 | 0 | + +### 3.3 `acq+docs`, hop off vs hop on + +``` +.venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop off \ + --overview-prefilter off --json-out eval/results/phase4_42_acqdocs_hop_off.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop on \ + --overview-prefilter off --json-out eval/results/phase4_42_acqdocs_hop_on.json +``` + +| arm | slice | n | R@5 | R@10 | R@20 | nDCG@10 (1st) | R@5 (fin) | R@10 (fin) | R@20 (fin) | **nDCG@10 (final)** | queries that hopped | +|---|---|---|---|---|---|---|---|---|---|---|---| +| hop **off** | all | 48 | 0.854 | 0.896 | 0.958 | 0.7194 | 0.854 | 0.896 | 0.958 | **0.7194** | 0 | +| hop **off** | `requires_crossref=true` | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | 0 | +| hop **off** | control (`=false`) | 13 | 0.769 | 0.769 | 0.846 | 0.7731 | 0.769 | 0.769 | 0.846 | **0.7731** | 0 | +| hop **on** | all | 48 | 0.854 | 0.896 | 0.958 | 0.7194 | 0.854 | 0.896 | 0.958 | **0.7194** | **7** | +| hop **on** | `requires_crossref=true` | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | **0** | +| hop **on** | control (`=false`) | 13 | 0.769 | 0.769 | 0.846 | 0.7731 | 0.769 | 0.769 | 0.846 | **0.7731** | **1** | + +Here the hop *does* fire โ€” 7 of 48 queries, 21 chunks added โ€” and the final +metrics still do not move. Hop precision, verbatim from the results JSON: + +``` +summary crossref_hop: { + "queries_with_hop": 7, + "fire_rate": 0.1458, + "chunks_added_total": 21, + "chunks_added_mean_when_fired": 3.0, + "hit_expected_source": 0, + "hopped_chunk_relevant": 0 +} +``` + +**0 of 7 hops landed in a gold source document, and 0 of 21 hopped chunks carried +gold text.** All 7 hops are `kind: "document"` title matches *between +`Documentation/*.md` files* (`triage_system.md โ†’ retrieval_pipeline.md`, +`prompt_inventory.md โ†’ verifier.md`, `architecture_overview.md โ†’ +indexing_pipeline.md`, โ€ฆ) โ€” i.e. they come from the distractor corpus, not from +the acquisition deal room. Not one hop originated on an `acq` PDF, consistent +with ยง3.2's "0 of 34 acq references resolve". + +Both `requires_crossref` rows and the 4.2 slice therefore hopped **zero** times +on `acq+docs` too: those queries already retrieve their answer document inside +the top 20 (`recall@10 = 1.000` on the slice), so the "not already represented" +guard is correct to suppress the hop โ€” there is nothing to fetch. + +`nDCG@10 (final)` is unchanged to 4 decimals in every arm above. The one query +whose final list grew and whose score could have moved, `acq_q12`, was already at +`nDCG@10 = 0.000` and the hop pulled `verifier.md`, which is unrelated: 0.000 โ†’ +0.000. **The hop neither helped nor hurt any measured number.** + +### 3.4 Sanity: hopped chunks really are in `documents`, tagged + +Direct pipeline call, `acq+docs`, `--crossref-hop on`, query `docs_d03`: + +``` +--- Performing hybrid retrieval for query: 'How many overviews does the overview router use?' on table 'eval_acq_plus_docs' --- +Retrieved 20 documents. +๐Ÿ”— Cross-reference hop: pulled 3 chunk(s) from 1 referenced document(s) (retrieval_pipeline.md). +first_stage: 20 documents: 23 +via_crossref chunks: 3 +{ + "chunk_id": "retrieval_pipeline.md_7", + "document_id": "retrieval_pipeline.md", + "via_crossref": true, + "crossref": { + "kind": "document", + "ref": "retrieval pipeline", + "from_chunk_id": "triage_system.md_2", + "from_document_id": "triage_system.md" + }, + "text_head": "7. Semantic cache ( agent/loop.py:130-154, 305-324, 587-594 ) Owned by Agent , not by the pipeline: TTLCache(maxsize=100, ttl=300) ( loop.py:33 ) keyed by raw " +} +first_stage carries via_crossref?: False +``` + +20 first-stage candidates โ†’ 23 final, 3 tagged `via_crossref: true`, and +`first_stage` untouched. Exactly the contract the pipeline documents. + +### 3.5 The rerank path still works, and `reranked` vs `final` line up + +The `rerank_error` detection had to change from list identity to element +identity (the hop rebuilds `documents` as a new list), so the rerank path was +re-run rather than assumed: + +``` +.venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop on \ + --overview-prefilter off --reranker BAAI/bge-reranker-v2-m3 \ + --json-out eval/results/phase4_42_acq_hop_on_rerank.json +``` + +``` + acq โ€” roadmap 4.2 slice: + requires_crossref=false n=13 recall@10=1.000 nDCG@10(1st)=0.863 | recall@10(fin)=1.000 nDCG@10(fin)=0.879 hops=0 + requires_crossref=true n=11 recall@10=1.000 nDCG@10(1st)=0.748 | recall@10(fin)=1.000 nDCG@10(fin)=0.822 hops=0 + +rerank_error: 0 +reranked==final on all: True +summary: ndcg@10_first_stage 0.8101 ndcg@10_reranked 0.8531 ndcg@10_final 0.8531 +``` + +`ndcg@10_reranked == ndcg@10_final` on every query, which is the correct +relationship when zero hops fired, and no query reported `rerank_error`. +(This arm is *not* part of the 4.2 A/B โ€” reranking is off in the shipped profile; +it is here only to prove the final metric composes with the rerank stage.) + +`--coverage-only` was also re-run (`acq`: `gold coverage 24/24 rows reachable`), +since the invariant check had to be skipped on that path. + +--- + +## 4. What the numbers say, and what they do not + +**The measurement infrastructure works** โ€” ยง3.4 shows the hop's output reaching +the scored list, ยง3.1 shows the metric is inert when nothing may change it, and +ยง3.3 shows the instrumentation catching a firing hop and correctly scoring it as +useless. + +**The 4.2 A/B is a negative result: no lift, no harm, because the hop cannot +reach the cases it was built for.** The blocker is not the hop, it is +index-time resolution. `rag_system/indexing/crossref.py::normalize_name` reduces +`08_regulatory_approval.pdf` to the literal string `08 regulatory approval`, and +the `document` family only matches when that whole string appears as whole words +in a chunk. The acquisition PDFs reference each other as *titles* ("Regulatory +Approval Documentation"), never with the numeric filename prefix, so the +`document` family matches nothing in that corpus. What the extractor *does* find +there โ€” `exhibit a`, `schedule 1`, `section 4.1` โ€” cannot resolve either, because +the Exhibits and Schedules are **sections inside** `01_acquisition_agreement.pdf`, +not separate files, and resolution is filename-based. + +Fixing that is an `rag_system/indexing/crossref.py` change (title-aware +resolution, or a numeric-prefix strip in `normalize_name`), which this wave does +not own and does not touch. Recording it as the finding is the deliverable. + +## 5. Caveats, stated plainly + +* **n = 11** on the `requires_crossref` slice, and it is at ceiling on recall + (1.000 at @5 on both `acq` and `acq+docs`). One query is 0.09 of any slice + figure. It cannot support a fine-grained adopt/reject call even if the hop + had fired. +* **`acq` is 13 chunks.** At the shipped `k = 20` the corpus is smaller than the + candidate budget, so *no* candidate-selection change can move a metric on it. + Any future 4.2 measurement needs either a bigger corpus or a smaller `k`, and + `--k 3` was tried here and still could not fire the hop. +* **The `acq+docs` full-set figures in this file (0.854 / 0.896 / 0.958, + nDCG@10 1st 0.7194) are lower than `BASELINE.md`'s Phase 4 baseline row + (0.917 / 0.958 / 1.000, 0.738).** That is the retry, not a regression: the + baseline row was run at `--retry profile` (on), everything here is `--retry + off` for determinism. Compare arms within this file, and compare the `mixed` + regression against the retry-off baseline only. +* **Eval nondeterminism**: with `--retry off`, no reranker and no decomposition, + this harness makes no LLM call on the query path, and all arms reused cached + indexes โ€” so the arms here are deterministic. Runs at `--retry profile` are + not (LLM query reformulation); the 0.887โ€“0.903 spread documented in + `phase4-crossref-prefilter.md` ยง2 is that effect. +* **Retrieval only.** No answer synthesis, no citation check. A hop that puts the + right document in the context but does not change `nDCG@10` would still be + invisible here โ€” the honest form of the "which document gets cited" gap + `BASELINE.md` already names. +* **`--overview-prefilter` is wired but unmeasured.** 4.3 needs document + overviews switched on at index time (an LLM call per document, which this + harness disables), so no 4.3 arm was run. The toggle exists so Wave 3 can run + one without a code edit. + +## 6. Files touched + +* `eval/run_eval.py` โ€” final-list metrics, invariant check, hop instrumentation, + `--crossref-hop` / `--overview-prefilter`. +* `eval/BASELINE.md` โ€” new subsection *Final-candidate-list metrics*. +* `eval/decisions/phase4-eval-final-metric.md` โ€” this file. + +Nothing under `rag_system/` or `Documentation/` was modified. diff --git a/eval/decisions/phase4-filters-askfolder.md b/eval/decisions/phase4-filters-askfolder.md new file mode 100644 index 00000000..e6a9455d --- /dev/null +++ b/eval/decisions/phase4-filters-askfolder.md @@ -0,0 +1,727 @@ +# Phase 4 items 4.4 and 4.6 โ€” metadata filter DSL and ephemeral "ask a folder" + +Date: 2026-08-09 +Status: **implemented; 4.4 is inert unless a caller passes `filters`, 4.6 is a new +CLI subcommand. Neither is on any gold-set metric, and neither can be.** +Owner of this wave: `rag_system/retrieval/`, `rag_system/api_server.py`, the +filters plumbing in `rag_system/agent/loop.py` and +`rag_system/pipelines/retrieval_pipeline.py`, and the **CLI section** of +`rag_system/main.py`. The `PIPELINE_CONFIGS` blocks, `Documentation/**` and +`backend/server.py` were deliberately **not** edited โ€” the config keys and diffs +they need are *proposed* at the bottom of this file and belong to the adoption +gate. + +Read the honesty section first: **there is no retrieval-quality evidence in this +document, because neither feature is a retrieval-quality change.** 4.4 changes +*what the caller is allowed to ask for*; when no filter is supplied it does +nothing at all, and that "nothing at all" is the strongest claim here โ€” it is +proven byte-for-byte below. 4.6 is packaging around the pipeline that already +ships. What follows is a mechanism proof and a non-regression proof, not a +benchmark. + +--- + +## 1. What shipped + +### 4.4 Metadata filter DSL + +**New module** โ€” `rag_system/retrieval/filters.py`. + +JSON, not a filter string. The roadmap's source surface is +`semantic_search(filters="field=value, field in (a,b)")`, a *string* that has to +be parsed; a parser is exactly the component an injection attack aims at. A JSON +object arrives already parsed, so the only work left is validation โ€” and +validation is the whole security story: + +```json +{"document_id": "07_nda.pdf"} +{"document_id": {"in": ["07_nda.pdf", "03_ip_certification.pdf"]}} +{"document_name": {"contains": "nda"}, "chunk_index": {"gte": 0, "lte": 4}} +``` + +| field | column | operators | +|-------|--------|-----------| +| `document_id` | `document_id` | `eq`, `in`, `contains` | +| `document_name` | `document_id` (substring) | `contains` only | +| `chunk_id` | `chunk_id` | `eq`, `in` | +| `chunk_index` | `chunk_index` | `eq`, `in`, `gt`, `gte`, `lt`, `lte` | + +Top-level keys are ANDed. A bare scalar is shorthand for `eq`, which is the +roadmap's `field=value`. There is no OR, no NOT, no nesting: nothing has asked +for them, and every operator is another string that ends up inside a SQL +predicate. + +`document_name` deserves its own line. It is **not** a column. Document ids are +the file's basename for anything indexed by the CLI and `<uuid>_<basename>` for +anything uploaded through the UI, so matching a name means substring-matching +the id โ€” hence `contains` and *only* `contains`. Offering `eq` there would +silently miss every UI-uploaded document, which is a trap rather than a feature. + +Security, following `rag_system/retrieval/document_fetch.py`'s precedent +(it refuses ids containing quotes/backslashes rather than escaping them): + +* **Refuse, don't escape.** A string value containing `'`, `"`, `\`, `;`, + backtick, or any control character raises `FilterError`. Nothing is repaired. +* The one exception is the `LIKE` metacharacters `%` and `_`, which *are* + escaped, with an explicit `ESCAPE '\'` clause, so that `contains` means + substring literally โ€” otherwise `contains: "01_acquisition"` would match + `01Xacquisition` and the filter would be quietly wider than it reads. +* Types are checked (`bool` is rejected for an int field even though Python + says `isinstance(True, int)`), strings are capped at 256 characters, IN-lists + at 256 items, integers at the int32 range the column actually holds. +* **Fail loud.** Unknown field, unknown operator, wrong type, empty IN-list and + the **empty object `{}`** all raise. `{}` in particular: a client bug that + sends an empty filter must not be indistinguishable from an unfiltered + search. Omitting the key entirely is how you ask for no filter. + +Compilation is deterministic โ€” fields are emitted in a fixed canonical order and +operators sorted, so the same filter object always produces the same +where-clause regardless of JSON key order. No LLM anywhere. (The roadmap's +LLM filter-*extraction* โ€” natural language to filter โ€” is explicitly "later"; +this is the deterministic layer it would target.) + +**Retriever** โ€” `rag_system/retrieval/retrievers.py`. + +`MultiVectorRetriever.retrieve()` gains a **keyword-only** `where=None`. It is +applied as a `prefilter=True` `.where()` on **both** legs โ€” the vector search +and the BM25/FTS search โ€” so a hybrid query cannot leak an excluded chunk in +through the lexical side. Prefiltering rather than post-filtering is the point: +post-filtering returns "up to k, minus whatever the filter removed", so a filter +for a rare document would come back empty while the document sat in the table. + +One deliberate behaviour change on the filtered path only: when `where` is set +and the search raises, the exception is **re-raised** instead of being swallowed +into `return []`. A filtered search that failed and a filtered search that +matched nothing are different answers, and only one of them is safe to show. + +**Pipeline** โ€” `rag_system/pipelines/retrieval_pipeline.py`. + +The compiled filter travels as a **thread-local scope** (`filter_scope()` / +`active_filter()`) opened by `run()`, plus a keyword-only `filters=` on +`retrieve_candidates()` for direct callers such as the eval harness. The +thread-local is not decoration: `EscalatingRetrievalPipeline` (roadmap 4.1) +overrides `retrieve_candidates` with a fixed four-argument signature and calls +`super()` **positionally**, so a new parameter that `run()` had to pass would +break it. A scope is invisible to that subclass, and the agent's parallel +sub-query fan-out enters `run()` *inside* each worker thread, so each worker +sets its own โ€” verified with a concurrency test below. + +Every path that reaches LanceDB is narrowed, not just the first stage: + +* `_first_stage` โ€” the main table and the late-chunk table. +* `_search_within_documents` โ€” ANDs the filter into the internal + `document_id IN (โ€ฆ)` clause, so the **cross-reference hop (4.2)** cannot be + used to reach a document the caller filtered out, and the **overview + prefilter's restrict mode (4.3)** stays inside the filter. +* `_get_surrounding_chunks_lancedb` โ€” context expansion. Without this, a + `chunk_index <= 0` filter would still pull chunk 1 back in as a neighbour and + "nothing that fails the filter reaches synthesis" would be false. + +When a filter is active, `retrieve_candidates` adds +`result["filters"] = {"spec": โ€ฆ, "where": โ€ฆ}` and emits a `filters_applied` +event. When there is no filter the result dict is unchanged โ€” no new key. + +**Agent** โ€” `rag_system/agent/loop.py` (plumbing only). + +`Agent.run(..., filters=โ€ฆ)` keyword-only; compiled once per user query and +handed to every retrieval that query performs, including each parallel +sub-query. Two non-obvious consequences, both deliberate: + +* **The semantic cache is filter-aware.** "What does the NDA say" and "what do + all ten documents say" have near-identical embeddings and different right + answers, so a cache entry now records its filter signature and only matches a + request with the same one. Both sides are `None` on the unfiltered path. +* **A filter skips triage**, the way `force_rag` does. Someone who filtered to a + named document has already decided the question is about the documents; + letting triage answer from general knowledge would ignore the filter and look, + from the outside, exactly like a filter that matched nothing. + +**API** โ€” `rag_system/api_server.py`. + +`/chat` and `/chat/stream` accept an optional `filters` object. It is compiled +in `_parse_chat_request`, so an invalid filter is a **400 before any retrieval +work happens**, and `filters` is only added to the `Agent.run` kwargs when +present โ€” an unfiltered request reaches the agent with exactly the arguments it +always did. + +**CLI** โ€” `python -m rag_system.main chat "<q>" --filters '<json>'`, same DSL, +same validation (bad JSON or an invalid filter exits 2 with the message). + +*Unrelated one-line fix in the same file*: `python -m rag_system.api_server +--port N` previously **accepted `--port` and ignored it**, silently listening on +8001. It now parses it. This was found by having a test talk to the wrong +server; leaving it would have been leaving a trap. + +### 4.6 Ephemeral "ask a folder" + +**New module** โ€” `rag_system/ask_folder.py`. **CLI** โ€” `rag_system/main.py`: + +``` +python -m rag_system.main ask <folder> "<question>" ["<question>" ...] + [--mode {fast,default}] [--interactive] [--agent] [--filters JSON] [--keep] +``` + +Index the folder's supported files into a throwaway LanceDB table, answer, delete +everything. No parallel pipeline: `IndexingPipeline` builds the index and the +standard `Agent` object answers โ€” via `agent.retrieval_pipeline.run()` by +default (the roadmap's "same pipeline, **no agent loop**") or via `agent.run(โ€ฆ, +force_rag=True)` under `--agent`, which adds decomposition and verification. + +Profile is `fast` per the roadmap, with two further switch-offs on top: +contextual enrichment (an LLM call per chunk) and **document overviews** (an LLM +call per document). Overviews only feed the agent's triage router and the +default-off overview prefilter, and neither runs here โ€” the answer path skips +triage because someone who typed `ask <folder>` has already decided the question +is about the folder. + +Cleanup is the feature, so it is arranged to be simple enough to be obviously +correct: **everything the run writes lives under one `tempfile.mkdtemp` +directory.** `storage.lancedb_uri`, `storage.db_path` and `overview_path` all +point inside it, so a single `rmtree` in a `finally` is the entire teardown, and +the directory is printed at the start and its removal reported at the end. +`--keep` skips deletion and says so loudly. + +`SIGTERM` is temporarily rebound to raise `SystemExit` so the `finally` actually +runs. This is not hypothetical: the first end-to-end run of this feature was +killed by a 2-minute harness timeout (`SIGTERM`) mid-synthesis and **leaked its +temp directory**, because Python's default SIGTERM disposition exits without +unwinding. `SIGINT` already unwound. `SIGKILL` cannot be caught and will still +leak a `localgpt-ask-*` directory under `$TMPDIR`; that is a property of +`kill -9`, and it is documented rather than papered over. + +--- + +## 2. Verification + +Everything below was executed on this machine today. Scratch scripts live in the +session scratchpad (not in `eval/`); paths are given so the runs can be redone. + +### 2.1 `py_compile` + +``` +$ .venv/bin/python -m py_compile rag_system/main.py rag_system/ask_folder.py \ + rag_system/retrieval/filters.py rag_system/retrieval/retrievers.py \ + rag_system/pipelines/retrieval_pipeline.py rag_system/agent/loop.py \ + rag_system/api_server.py +py_compile OK (7 files) +``` + +`npx tsc --noEmit` was **not** run: no file under `src/` was touched. The +frontend sends no `filters` and is unaffected. + +### 2.2 No-filter behaviour is byte-identical to pre-change + +The claim that matters most, so it was measured rather than argued. + +A pre-change copy of the tree was reconstructed in a scratch directory: +`retrievers.py` straight from `git HEAD` (it carried no working-tree changes +before this wave), and `retrieval_pipeline.py` as the current file with **this +wave's edits reversed one by one, every reversal asserting that its target was +present** (`scratchpad/make_baseline.py`; it also asserts the result mentions +none of `filter_scope`, `compile_filters`, `active_filter`, `filter_where`, and +then compiles it). `git HEAD` is not a usable baseline for +`retrieval_pipeline.py` because wave 1's cross-reference/overview-prefilter work +is in the working tree and not in HEAD. + +Both trees then ran the same five queries through +`RetrievalPipeline.retrieve_candidates()` on the existing eval index +`eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq`, table `eval_acq` +(13 chunks, 10 PDFs), dumping per-candidate `chunk_id`, `document_id`, +`chunk_index`, `score`, `_distance`, `bm25`, text length and a text hash. The +evidence-sufficiency retry was forced off, because it calls an LLM to reformulate +and the two arms would not be comparable otherwise. + +``` +$ cmp candidates_baseline.json candidates_after.json +IDENTICAL: candidates_baseline.json == candidates_after.json +$ md5 -q candidates_baseline.json candidates_after.json +63e4edbcf805bdf87498c6c26fb36541 +63e4edbcf805bdf87498c6c26fb36541 +$ wc -c candidates_baseline.json candidates_after.json + 41602 candidates_baseline.json + 41602 candidates_after.json +``` + +**Caveat, stated plainly**: this is five queries on one 13-chunk index with the +retry off, not the gold set. It proves the no-filter code path is unchanged; the +code-level argument is what generalises it โ€” `compile_filters(None)` returns +`None`, the scope is never entered, `active_filter()` returns `None`, +`where=None` makes the retriever's `_filtered()` helper the identity, and +`combine(x, None)` returns `x`. Every changed expression reduces to the original. + +### 2.3 Filter DSL unit tests โ€” 36/36 + +`scratchpad/test_filters_unit.py`. Full verbatim output is in the session log; +the compiled forms and a sample of the refusals: + +``` + OK document_id scalar eq -> document_id = '07_nda.pdf' + OK document_id in-list -> document_id IN ('07_nda.pdf', '03_ip_certification.pdf') + OK document_name contains -> document_id LIKE '%01\_acquisition%' ESCAPE '\' + OK chunk_index range -> chunk_index >= 0 AND chunk_index <= 4 + OK multi-field AND -> document_id IN ('a.pdf', 'b.pdf') AND chunk_index <= 0 + key order A -> document_id = 'x.pdf' AND chunk_index >= 1 AND chunk_index <= 3 + key order B -> document_id = 'x.pdf' AND chunk_index >= 1 AND chunk_index <= 3 + OK identical regardless of JSON key order + + OK single quote in value -> FilterError: filters.document_id.eq contains a forbidden character ("'"); quoting characters are refused, not escaped. + OK semicolon / stacked statement -> FilterError: filters.document_id.eq contains a forbidden character (';'); ... + OK backslash -> FilterError: filters.document_id.eq contains a forbidden character ('\\'); ... + OK unknown field -> FilterError: Unsupported filter field(s): page. Supported: document_id (eq, in, contains); document_name (contains); chunk_id (eq, in); chunk_index (eq, in, gt, gte, lt, lte). + OK bool as chunk_index -> FilterError: filters.chunk_index.eq must be an integer, got bool. + OK empty object -> FilterError: filters is empty. Omit the field entirely to search without a filter; ... + +36 passed, 0 failed +``` + +### 2.4 Filtered retrieval on the real eval index โ€” 37/37 + +`scratchpad/test_filters_pipeline.py`, same `eval_acq` index. Verbatim: + +``` +== 1. unfiltered baseline (what the filter has to change) == + hybrid 13 chunks over 10 documents: [...all ten PDFs...] + vector_only 13 chunks over 10 documents: [...all ten PDFs...] + fts_only 10 chunks over 10 documents: [...all ten PDFs...] + [PASS] hybrid: unfiltered result has no 'filters' key โ€” keys=['documents', 'first_stage', 'query_used', 'retry'] + +== 2. document_id equality restricts BOTH legs == + [PASS] hybrid: only 07_nda.pdf returned โ€” 1 chunk(s), documents=['07_nda.pdf'] + [PASS] vector_only: only 07_nda.pdf returned โ€” 1 chunk(s), documents=['07_nda.pdf'] + [PASS] fts_only: only 07_nda.pdf returned โ€” 1 chunk(s), documents=['07_nda.pdf'] + [PASS] hybrid: result carries the applied where-clause โ€” {'spec': {'document_id': '07_nda.pdf'}, 'where': "document_id = '07_nda.pdf'"} + +== 2b. the raw retriever, one leg at a time == + [PASS] MultiVectorRetriever fts_only where=... โ†’ only the NDA โ€” 1 row(s) filtered vs 10 unfiltered over 10 documents + [PASS] MultiVectorRetriever vector_only where=... โ†’ only the NDA โ€” 1 row(s) filtered vs 13 unfiltered over 10 documents + +== 3. other operators == + [PASS] IN-list returns exactly those two documents โ€” ['03_ip_certification.pdf', '07_nda.pdf'] + [PASS] document_name contains 'nda' โ†’ 07_nda.pdf โ€” ['07_nda.pdf'] + [PASS] underscore is literal, not a wildcard โ€” ['01_acquisition_agreement.pdf'] + [PASS] chunk_index <= 0 keeps only first chunks โ€” chunk_index values [0] + [PASS] two fields AND together โ€” [('01_acquisition_agreement.pdf', 1)] + +== 4. a filter that matches nothing does NOT fall back == + [PASS] empty result, not an unfiltered one โ€” 0 chunk(s) + +== 5. invalid filters fail loud at the pipeline boundary == + [PASS] unknown field rejected / unknown operator rejected / empty object rejected / + wrong type rejected / injection: quote rejected / injection: stacked rejected / + injection: backslash rejected (all FilterError, messages as in ยง2.3) + +== 5b. the table is intact after every injection attempt == + [PASS] eval_acq still has 13 rows โ€” 13 rows + +== 6. pipeline.run(filters=...) validates before doing any work == + [PASS] run() rejects an invalid filter โ€” FilterError: Unsupported filter field(s): bogus. ... + +== 7. internally-scoped searches are narrowed too (crossref hop / restrict mode) == + [PASS] _search_within_documents honours the active filter โ€” scoped=[0] unscoped=[0, 1] + +== 8. concurrent queries with different filters do not cross-wire == + [PASS] thread A saw only its own filter โ€” ['07_nda.pdf'] + [PASS] thread B saw only its own filter โ€” ['03_ip_certification.pdf'] + [PASS] unfiltered thread was unaffected โ€” [...all ten PDFs...] + [PASS] no filter leaked into the next unfiltered query โ€” keys=['documents', 'first_stage', 'query_used', 'retry'] + +37/37 assertions passed +``` + +Note on ยง2 and ยง2b: with `retrieval_k=20` against a 13-chunk index, "1 chunk" +*is* the whole of `07_nda.pdf` โ€” the NDA is a single chunk. The +`1 filtered vs 13 unfiltered` comparison is what makes the assertion +non-vacuous. + +### 2.5 HTTP end-to-end โ€” `/chat` and `/chat/stream` + +`scratchpad/test_filters_http.py` starts the RAG API as a child process against +the same `eval_acq` index (`LANCEDB_PATH`) and a throwaway chat DB, then drives +it over HTTP. **21/21 assertions passed.** The nine malformed/injecting filters +were each sent to *both* endpoints: + +``` + rag-api healthy on 8011 + +== malformed filters are 400, on both endpoints == + [PASS] /chat unknown field โ†’ 400 โ€” status=400 body={ "error": "Invalid filters: Unsupported filter field(s): page. Supported: document_id (eq, in, contains); document_name (contains); chunk_id (eq, in); chunk_index (eq, in, gt, gโ€ฆ + [PASS] /chat unknown operator โ†’ 400 โ€” status=400 body={ "error": "Invalid filters: filters.document_id does not support the 'regex' operator. Supported: eq, in, contains." } + [PASS] /chat empty object โ†’ 400 โ€” status=400 body={ "error": "Invalid filters: filters is empty. Omit the field entirely to search without a filter; an empty filter object is refused so that a client bug cannot look like an unfiโ€ฆ + [PASS] /chat wrong type โ†’ 400 โ€” status=400 body={ "error": "Invalid filters: filters.chunk_index.eq must be an integer, got str." } + [PASS] /chat not an object โ†’ 400 โ€” status=400 body={ "error": "Invalid filters: filters must be a JSON object, got str. Supported fields: โ€ฆ" } + [PASS] /chat injection: quote โ†’ 400 โ€” status=400 body={ "error": "Invalid filters: filters.document_id.eq contains a forbidden character (\"'\"); quoting characters are refused, not escaped." } + [PASS] /chat injection: stacked statement โ†’ 400 โ€” status=400 body={ "error": "Invalid filters: filters.document_id.eq contains a forbidden character (';'); quoting characters are refused, not escaped." } + [PASS] /chat injection: backslash โ†’ 400 โ€” status=400 body={ "error": "Invalid filters: filters.document_id.eq contains a forbidden character ('\\\\'); quoting characters are refused, not escaped." } + [PASS] /chat injection: LIKE wildcard smuggling via contains โ†’ 400 โ€” status=400 body={ "error": "Invalid filters: filters.document_name.contains contains a forbidden character (\"'\"); quoting characters are refused, not escaped." } + โ€ฆ the same nine, verbatim, against /chat/stream: all [PASS] 400. + +== a valid filter restricts a real answer == + [PASS] /chat with a valid filter returns 200 โ€” 200 + [PASS] every cited chunk is from the filtered document โ€” 1 source(s) from ['07_nda.pdf'] in 59.9s + + answer: Answer: +According to the Mutual Non-Disclosure Agreement ("NDA") entered into as of October 1, 2024, by TechCorp Industries, Inc. (located at 500 Technology Drive, San Francisco, CA 94105) and StartupXYZ LLC (located at 123 Innovation Way, Palo Alto, CA 94301), the confidentiality obligations are explicitly defined in Section 2 of the document ("OBLIGATIONS"). Each Party agrees to adhere to the foโ€ฆ + +== the same question unfiltered cites other documents == + [PASS] unfiltered answer spans more than one document โ€” ['01_acquisition_agreement.pdf', '07_nda.pdf', '10_closing_checklist.pdf'] + +21/21 assertions passed +``` + +The last pair is the one that matters: same question, same index, `retrieval_k=3` +โ€” filtered cites one document, unfiltered cites three. + +### 2.6 4.6 end-to-end on `eval/corpora/acquisition/` (10 PDFs) + +One real question, one real answer, exit code 0. + +``` +$ .venv/bin/python -m rag_system.main ask eval/corpora/acquisition \ + "What is the total purchase price of the acquisition and how is it broken down?" + +๐Ÿ“‚ ask: 10 file(s) from /Users/prompt/videos/localgpt_08082026/localGPT/eval/corpora/acquisition +๐Ÿ—‘๏ธ ephemeral index: table 'ask_d74a23299d65' in /var/folders/โ€ฆ/T/localgpt-ask-jf_scljx (profile 'fast') +โฑ๏ธ indexed in 9.4s +``` + +Answer (tail; the corpus's planted numbers are `$45,000,000` original, +`$43,330,000` adjusted, and a `$39,630,000` total at closing): + +``` +โ€ฆt Adjustment ($175,000); Revenue Recognition Impact (Implied value adjustment at 15x) ($1,275,000). +* Adjusted Purchase Price: **$43,330,000**. + +Section 6 of the Financial Adjustments Memo provides a revised payment structure reflecting these +adjustments ("As revised from Document: Acquisition Agreement Section 2.2"): +* (a) Cash at closing: **$28,330,000** (adjusted); +* (b) Stock consideration: **$10,000,000**; and +* (c) Earnout payments: **$5,000,000**. + +Additionally, a Contingent Liability Reserve of **$1,300,000** was recommended to be held in escrow +per Exhibit C - Earnout Terms. โ€ฆ + +The **Closing Checklist** indicates a "Total at Closing" of **$39,630,000**, which corresponds to the +sum of Cash ($28,330,000) + Escrow deposit ($1,300,000) + Stock issuance ($10,000,000). The $5,000,000 +earnout payments are treated as future contingent consideration rather than immediate cash required at +closing. +``` + +Citations and teardown: + +``` +๐Ÿ“Ž Sources (452.7s): + [1] 05_financial_adjustments.pdf#0 (score 0.5329) + FINANCIAL ADJUSTMENTS MEMO FINANCIAL ADJUSTMENTS MEMORANDUM To: Deal Team From: Finance Department Date: December 23, 2024 Re: Purchase Price Adjustments - StartupXYZ Acquisition Following our review in connection with tโ€ฆ + [2] 01_acquisition_agreement.pdf#0 (score 0.5259) + ACQUISITION AGREEMENT ACQUISITION AGREEMENT This Acquisition Agreement ("Agreement") is entered into as of January 15, 2025, by and between TechCorp Industries, Inc. ("Buyer") and StartupXYZ LLC ("Seller"). ARTICLE I - Dโ€ฆ + [3] 10_closing_checklist.pdf#0 (score 0.5086) + CLOSING CHECKLIST Acquisition of StartupXYZ LLC by TechCorp Industries, Inc. Closing Date: March 1, 2025 Closing Location: Wilson & Partners LLP, San Francisco I. PRE-CLOSING CONDITIONS A. Regulatory [X] HSR Filing submiโ€ฆ + [4] 02_due_diligence_report.pdf#0 (score 0.4959) + DUE DILIGENCE REPORT CONFIDENTIAL DUE DILIGENCE REPORT Prepared for: TechCorp Industries, Inc. Subject: StartupXYZ LLC Date: December 20, 2024 Prepared by: Morrison & Associates, LLP EXECUTIVE SUMMARY This report summariโ€ฆ + [5] 04_risk_assessment.pdf#0 (score 0.4910) + RISK ASSESSMENT MEMO CONFIDENTIAL RISK ASSESSMENT MEMORANDUM To: TechCorp Board of Directors From: Corporate Development Team Date: December 22, 2024 Re: Risk Assessment - StartupXYZ Acquisition This memo summarizes key โ€ฆ + โ€ฆ and 5 more source chunk(s) + +๐Ÿงน Removed the ephemeral index (/var/folders/โ€ฆ/T/localgpt-ask-jf_scljx). +โฑ๏ธ total 462.1s +``` + +**Cleanup, before and after.** Both listings were taken with the same command; +`lancedb/` and `index_store/` are identical: + +``` +before after +--- lancedb/ --- --- lancedb/ --- +text_pages_82b2f5a9-โ€ฆ.lance text_pages_82b2f5a9-โ€ฆ.lance +--- index_store/ --- --- index_store/ --- +overviews overviews +index_store/overviews: index_store/overviews: +2fb7a91a-โ€ฆ.jsonl 2fb7a91a-โ€ฆ.jsonl +66ac9551-โ€ฆ.jsonl 66ac9551-โ€ฆ.jsonl +82b2f5a9-โ€ฆ.jsonl 82b2f5a9-โ€ฆ.jsonl +``` + +`diff` reports **no difference** in those two sections. No `ask_*` table, no new +overview JSONL, no `.vectors.npz`. + +**The one leak, reported because it happened.** An *earlier* attempt at this same +run was killed by a 2-minute harness timeout and left +`$TMPDIR/localgpt-ask-syr5n_ob` behind (156K, containing only an empty `lancedb` +directory). That is what prompted the SIGTERM handler described in ยง1. It was +then verified: + +``` +$ .venv/bin/python -m rag_system.main ask eval/corpora/acquisition "test" & # then SIGTERM mid-index +temp dir: /var/folders/โ€ฆ/T/localgpt-ask-_smy40vj +exists before SIGTERM: yes +exit=143 +removed after SIGTERM +๐Ÿงน Removed the ephemeral index (/var/folders/โ€ฆ/T/localgpt-ask-_smy40vj). +โฑ๏ธ total 13.3s +``` + +The stale `localgpt-ask-syr5n_ob` from the pre-fix run was deleted by hand. +`ls -d $TMPDIR/localgpt-*` now lists nothing but the running smoke test's own +directory. + +### 2.7 The other `ask` branches, on a two-file scratch folder + +The run above only exercises the default path, so the three remaining branches +were run on a two-document scratch folder (`widget_spec.md`: torque 47 Nm, +service interval 900 hours, bearing BR-2210; `pricing.md`: $3,150/unit, 12% +above 50 units). All four exited as intended and all three indexing runs cleaned +up. Output below is filtered to the mode's own lines (the full multi-line answers +are longer than what the filter kept). + +``` +########## A: --agent ########## +โฑ๏ธ indexed in 3.5s +โ“ What is the service interval and the list price of the Widget X-9? +๐Ÿ’ฌ Answer: โ€ฆ The list price of the Widget X-9 is 3,150 dollars per unit, and volume orders + above 50 units receive a 12 percent discount. Regarding maintenance requirements, its + service interval is specified as 900 operating hours. +๐Ÿ“Ž Sources (69.4s): [1] pricing.md#0 (score 0.6490) [2] widget_spec.md#0 (score 0.6070) +๐Ÿงน Removed the ephemeral index (โ€ฆ/localgpt-ask-2v0ldoxa). +โฑ๏ธ total 72.9s +exitA=0 + +########## B: --filters '{"document_id": "pricing.md"}' ########## +โฑ๏ธ indexed in 3.1s +๐Ÿ”Ž Metadata filter (prefilter, both legs): document_id = 'pricing.md' +๐Ÿ”Ž Metadata filter applied: document_id = 'pricing.md' โ†’ 1 candidate(s). +๐Ÿ“Ž Sources (67.7s): [1] pricing.md#0 (score 0.5035) โ† widget_spec.md excluded +๐Ÿงน Removed the ephemeral index (โ€ฆ/localgpt-ask-i9ewqvun). +exitB=0 + +########## C: --interactive (two questions piped on stdin, then a blank line) ########## +โ“ What is the replacement bearing part number? +๐Ÿ’ฌ โ€ฆ the replacement bearing part number for the Widget X-9 is BR-2210. +โ“ What discount applies above 50 units? +๐Ÿ’ฌ โ€ฆ volume orders for the Widget X-9 that exceed **50 units** receive a **12 percent discount**. +โ“ Ask a follow-up (blank line to finish): +๐Ÿงน Removed the ephemeral index (โ€ฆ/localgpt-ask-4jzkp3tg). +โฑ๏ธ total 141.4s +exitC=0 + +########## D: bad filter ########## +โŒ Invalid --filters: Unsupported filter field(s): page. Supported: document_id (eq, in, + contains); document_name (contains); chunk_id (eq, in); chunk_index (eq, in, gt, gte, lt, lte). +exitD=2 +``` + +B is the interesting one: the filter is honoured on an index that was created +seconds earlier, and `widget_spec.md` โ€” which contains the torque and bearing +facts and would otherwise be candidate 2 โ€” never enters the context. + +Both filters (4.4) and ask (4.6) therefore compose, and 4.4 works against an +ephemeral table as well as a persistent one. + +**After all of the above**: `ls -d $TMPDIR/localgpt-*` lists nothing, `lancedb/` +holds only the single pre-existing `text_pages_82b2f5a9-โ€ฆ.lance`, and +`index_store/overviews/` holds only the same three pre-existing `.jsonl` files. + +### 2.8 Smoke test + +``` +$ .venv/bin/python eval/smoke_e2e.py + [PASS] both services became healthy + [PASS] upload accepted the PDF + [PASS] index build returned 200 with no error + [PASS] index linked to session + [PASS] q1: planted fact '9.2' in answer [PASS] q1: source_documents non-empty + [PASS] q1: [Confidence: N%] tag present โ€” tag=[Confidence: 100%] + [PASS] q1: message_count == 2 โ€” got 2 + [PASS] q2: planted fact 'TS-71' in answer [PASS] q2: source_documents non-empty + [PASS] q2: [Confidence: N%] tag present โ€” tag=[Confidence: 100%] + [PASS] q2: message_count == 4 โ€” got 4 + [PASS] q3: planted fact '36' in answer [PASS] q3: source_documents non-empty + [PASS] q3: [Confidence: N%] tag present โ€” tag=[Confidence: 100%] + [PASS] q3: message_count == 6 โ€” got 6 + [PASS] q4: planted fact 'drip tray' in answer [PASS] q4: source_documents non-empty + [PASS] q4: [Confidence: N%] tag present โ€” tag=[Confidence: 100%] + [PASS] q4: message_count == 8 โ€” got 8 + [PASS] messages/save returned 200 + [PASS] saved assistant message round-trips out of SQLite โ€” 10 messages in session + [PASS] saved source_documents round-trip in metadata + [PASS] saved steps round-trip in metadata + [PASS] final message_count == 10 โ€” got 10 + wall clock 280.6s +25/25 assertions passed +exit=0 +``` + +(The four `[PASS] qN: โ€ฆ` lines are reflowed two-per-line here purely to fit; +the wording is verbatim.) + +--- + +## 3. Limitations โ€” read before enabling anything + +* **Only real columns are filterable.** The LanceDB text table has six columns + (`vector`, `text`, `chunk_id`, `document_id`, `chunk_index`, `metadata`) and + `metadata` is a **JSON string**. The roadmap names "document name, page, + date"; page and date live inside that JSON string and are therefore **not + filterable**. They were not faked with a `metadata LIKE '%โ€ฆ%'` substring + match, which would look like it worked and quietly be wrong (`"page": 3` + matches `"page": 31`, and any user text containing the phrase matches too). + Fixing this needs page/date promoted to real columns โ€” a schema change and a + re-index. See the backlog. +* **`document_name` is a substring of `document_id`.** For UI-uploaded + documents the id is `<uuid>_<name>`, so `contains: "report"` also matches a + document whose *uuid* happens to contain "report" (unlikely) and matches every + document whose name contains it (intended). There is no exact name match. +* **No OR / NOT / nesting.** `{"document_id": {"in": [...]}}` covers the common + disjunction. Anything else is a new feature, not a config. +* **No gold-set number exists for 4.4 and one cannot be produced today**: every + gold row is scored against an unfiltered corpus, so a filter can only ever + make recall worse there. Measuring filters needs gold rows that *carry* a + filter and an expected in-scope answer. Backlog. +* **4.6 is slow on document-sized chunks.** The one real run below spent + **452.7s of its 462.1s in synthesis** โ€” the `fast` profile's `retrieval_k=10` + against ten single-chunk documents hands the generation model ten whole + documents. Indexing was 9.4s. This is not a regression (it is the shipped + pipeline behaving normally at this chunk size) but "fast profile" reads like a + promise this mode does not keep. +* **4.6 has no eval coverage at all.** It is a CLI wrapper; the harness does not + drive CLI subcommands. +* **`SIGKILL` still leaks the temp directory.** Nothing can be done about that + from inside the process. +* **`ask --agent` barely differs from the default on the `fast` profile.** `fast` + has `query_decomposition` and `verification` off, so on that profile the flag + buys only the agent's history/caching bookkeeping โ€” run A in ยง2.7 produces no + `[Confidence: N%]` tag for exactly that reason. `--agent --mode default` is + where the flag means something, and that combination has **not** been run. +* **`--interactive` was verified with piped stdin, not a TTY.** The prompt string + and the pipeline's log lines interleave on one line in that mode, which is + cosmetic but ugly; on a real terminal it reads normally. + +--- + +## 4. Proposed config keys (for `rag_system/main.py`, at the adoption gate) + +**None are required.** 4.4 has no flag by design: no `filters` argument means +byte-identical behaviour, so there is nothing to switch off, and a flag would +only create a state where a caller's explicit filter is silently ignored โ€” the +exact failure this feature must never have. 4.6 is a CLI subcommand, opt-in by +existing. + +Two keys are *available* if the gate wants them discoverable rather than +hard-coded in `rag_system/ask_folder.py`: + +```python + # Optional. "ask" (roadmap 4.6) currently hard-codes these on top of the + # `fast` profile; a profile block would let a user change them. + "ask": { + "profile": "fast", # profile the ephemeral index is built with + "enrichment": False, # LLM call per chunk + "overviews": False # LLM call per document; nothing here reads them + } +``` + +If the gate would rather cap filter surface centrally, the limits currently +living as module constants in `rag_system/retrieval/filters.py` +(`_MAX_STRING_LENGTH = 256`, `_MAX_IN_ITEMS = 256`) are the candidates. They are +guard rails, not tuning knobs, which is why they are constants today. + +## 5. Proposed gateway diff (`backend/server.py`, **not applied** โ€” not my file) + +The RAG API accepts `filters` today; the Node/Python gateway at +`backend/server.py` drops unknown keys, so a browser client cannot yet reach it. +Two changes: + +```python +# CHAT_OPTIONS, next to "retrieval_mode": + "filters": (dict, ()), +``` + +`normalize_options` calls `caster(data[key])`; `dict({...})` copies a dict and +raises `TypeError`/`ValueError` on anything else, which the existing +`except (TypeError, ValueError)` branch already passes through unchanged โ€” so a +non-object `filters` still reaches the RAG API and is rejected there with the +400 that carries the real message. No validation logic is duplicated in the +gateway, deliberately: one validator, one error text. + +```python +# should_use_rag(...), alongside the force_rag short-circuit: + if options.get("filters"): + return True # an explicit filter is a statement that this is a document question +``` + +(the second is the gateway-level twin of the agent-side rule in ยง1). + +## 6. Proposed documentation diffs (none applied this wave) + +* **`Documentation/retrieval_pipeline.md`** + * New section *Metadata filters*: the JSON DSL table (field โ†’ column โ†’ + operators), that both legs are prefiltered, that internal + document-scoped searches and context expansion are narrowed too, the + `filters_applied` SSE event and the `result["filters"]` shape, and the + explicit statement that an absent `filters` key changes nothing. + * A line under the retry: the filter is re-applied to the reformulated query + because it is read from the scope, not passed per call. +* **`Documentation/api.md` / whichever file documents `/chat`** โ€” the optional + `filters` body field, the 400 contract, and the fact that a filter implies + `force_rag`. +* **`Documentation/design_rationale.md`** + * ยง4 or a new short section: why the filter language is JSON rather than the + source project's filter *string* (a parser is the attack surface), and why + values are refused rather than escaped (`document_fetch.py`'s precedent). + * Why `document_name` is `contains`-only. + * Why the filter travels as a thread-local scope rather than a parameter (the + 4.1 subclass's positional `super()` call) โ€” this is a real coupling and the + next person to touch `retrieve_candidates` needs to know it exists. +* **`Documentation/research_roadmap.md`** โ€” Phase 4 rows 4.4 and 4.6: mark + implemented, pointing here; note that 4.4 covers `document_id` / + `document_name` / `chunk_index` / `chunk_id` but **not** page or date, which + the row currently promises. +* **`README.md` / CLI docs** โ€” the `ask` subcommand and `chat --filters`. + +## 7. Couplings the next person needs to know + +1. **`retrieve_candidates`'s positional signature is load-bearing.** + `EscalatingRetrievalPipeline` (roadmap 4.1) overrides it and calls `super()` + with four positional arguments. That is why `filters` is keyword-only and why + `run()` uses a thread-local scope instead of passing it down. Anyone adding + another parameter must do the same, or update + `rag_system/agent/escalation.py` in the same change. +2. **`retrieve_candidates`'s body moved into `_retrieve_candidates_filtered`, + and `run()`'s second half into `_run_after_candidates`.** Both are pure + extractions with the filter scope wrapped around them; no logic moved between + them. A merge with another change to either method will conflict textually + and should be resolved by keeping the split. +3. **`_post_candidates` is still the single tail hook** for every + `retrieve_candidates` exit path (wave 1's contract). Nothing here changed it. +4. **The semantic cache entry gained a `"filters"` key.** Entries written by an + older build have no such key, so `dict.get` returns `None`, which matches an + unfiltered request โ€” the intended behaviour, and the reason the comparison is + `.get(...) != signature` rather than a lookup that would `KeyError`. +5. **`rag_system/api_server.py`'s `__main__` now parses `--port`.** Anything that + started it with `--port` and relied on getting 8001 anyway will now get the + port it asked for. Nothing in the repo does; `eval/smoke_e2e.py` starts it + without `--port`. + +## 8. Backlog + +1. **Gold rows that carry filters.** ~6 rows of the form + *(query, filters, expected)* where the expected answer is inside the filtered + scope and a plausible distractor is outside it. Without these, 4.4 has no + number and cannot get one. The harness would need `retrieve_candidates(..., + filters=โ€ฆ)`, which is already the public signature. +2. **Promote `page` (and any date) to real LanceDB columns** so the roadmap's + full 4.4 surface is reachable. Schema change + re-index; the DSL grows two + rows in its field table and nothing else. +3. **A harm check for filters + the 4.2 hop.** The hop is currently narrowed by + the filter, which is right, but it means a filtered query can silently lose + the hop's benefit. Worth a line in the docs once 4.2 is benchmarked at all. +4. **LLM filter extraction** (the roadmap's "later"): natural language โ†’ + this DSL, on the enrichment model, with the compiled where-clause shown to + the user before it runs. The deterministic layer it needs now exists; the + measured ~0.999 F1 claim in the roadmap is for easy/medium translations and + should be re-measured on this DSL before anything ships. +5. **`ask --agent --mode default`** has never been run; it is the only untested + combination of the new CLI flags. Cheap to cover once someone has 10 minutes + of generation budget. +6. **Reconsider `ask`'s default `retrieval_k`.** Ten whole documents into one + synthesis prompt is what made the run below take 7.7 minutes; a lower `k` + (or chunking that produces more, smaller chunks) is the fix, but it is a + quality/latency trade and should be measured, not guessed. diff --git a/eval/decisions/phase4-retrieval-benchmarks.md b/eval/decisions/phase4-retrieval-benchmarks.md new file mode 100644 index 00000000..281f8248 --- /dev/null +++ b/eval/decisions/phase4-retrieval-benchmarks.md @@ -0,0 +1,741 @@ +# Phase 4 items 4.2 and 4.3 โ€” the retrieval benchmark matrix, on rebuilt indexes + +Date: 2026-08-09 +Status: **measured.** Both mechanisms now fire and both move numbers. Neither +result is a clean win. A PROPOSED call is recorded per item at the end; **the +adoption gate makes the final call, not this document.** + +Scope of this wave: `eval/.eval_indexes/**` (rebuilds), `eval/results/**`, +`eval/run_eval.py` (one additive flag, ยง3.1), `eval/BASELINE.md`, this file. +Nothing under `rag_system/`, `Documentation/`, `src/` or `backend/` was +modified. One `rag_system` design finding is recorded in ยง2.5 with a proposed +diff and deliberately **not** applied. + +Prior art this builds on, and does not repeat: +[`phase4-crossref-prefilter.md`](phase4-crossref-prefilter.md) (what shipped, +plus the *Gate correction (2026-08-09)* appendix that fixed the resolver) and +[`phase4-eval-final-metric.md`](phase4-eval-final-metric.md) (the final-list +metric, and the first A/B โ€” a zero-hop negative result on indexes built before +the resolver fix). + +**Determinism protocol, applied to every run below:** `--retry off` (the +evidence-sufficiency retry is an LLM reformulation and is nondeterministic), +reranker off, no empty-string env vars, no `Documentation/` edit between arms. +With those settings the query path makes **no LLM call at all**, so each pair of +arms differs only in the flag under test. + +**Latency is not reported and no performance claim is made.** A second agent was +running answer-quality benchmarks against the same Ollama instance throughout; +every wall-clock number in this wave is contended and meaningless. + +--- + +## 1. Index rebuild and re-baseline + +### 1.1 Why + +`rag_system/indexing/crossref.py` was changed at the gate (numeric-prefix-stripped +filename aliases, so `08_regulatory_approval.pdf` also registers as +`"regulatory approval"`). Cross-references are stamped into chunk metadata **at +index time**, so the existing `acq` / `acq_plus_docs` eval indexes โ€” built before +the extractor existed, and certainly before the fix โ€” carried none of it. Both +were deleted and rebuilt: + +``` +rm -rf eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq \ + eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq_plus_docs + +.venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop off \ + --overview-prefilter off \ + --json-out eval/results/phase4_w3_rebuild_acq_hop_off.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop off \ + --overview-prefilter off \ + --json-out eval/results/phase4_w3_rebuild_acqdocs_hop_off.json +``` + +Build-time log lines, verbatim: + +``` +acq ๐Ÿ”— Cross-references: 68 reference(s) in 11 chunk(s); 34 resolved to 9 document(s). +acq+docs ๐Ÿ”— Cross-references: 215 reference(s) in 70 chunk(s); 93 resolved to 21 document(s). +``` + +### 1.2 Verification, read straight out of LanceDB + +Not from the build log โ€” from the built table, decoding the `metadata` column +and reading `["metadata"]["crossrefs"]` (`VectorIndexer` stores +`json.dumps(chunk)` there, so the real metadata is nested one level down). + +``` +LanceDB connection established at: eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq +eval_acq: 13 chunks +chunks carrying crossrefs: 11 total refs: 68 resolved: 34 distinct target documents: 9 + +resolved edges (source_doc -> target_doc via ref): + 01_acquisition_agreement.pdf --[document: due diligence report]--> 02_due_diligence_report.pdf + 02_due_diligence_report.pdf --[document: acquisition agreement]--> 01_acquisition_agreement.pdf + 02_due_diligence_report.pdf --[document: customer consents]--> 09_customer_consents.pdf + 02_due_diligence_report.pdf --[document: financial adjustments]--> 05_financial_adjustments.pdf + 02_due_diligence_report.pdf --[document: regulatory approval]--> 08_regulatory_approval.pdf + 03_ip_certification.pdf --[document: acquisition agreement]--> 01_acquisition_agreement.pdf + 03_ip_certification.pdf --[document: legal opinion]--> 06_legal_opinion.pdf + 03_ip_certification.pdf --[document: risk assessment]--> 04_risk_assessment.pdf + 04_risk_assessment.pdf --[document: acquisition agreement]--> 01_acquisition_agreement.pdf + 04_risk_assessment.pdf --[document: closing checklist]--> 10_closing_checklist.pdf + 04_risk_assessment.pdf --[document: due diligence report]--> 02_due_diligence_report.pdf + 04_risk_assessment.pdf --[document: financial adjustments]--> 05_financial_adjustments.pdf + 04_risk_assessment.pdf --[document: regulatory approval]--> 08_regulatory_approval.pdf + 05_financial_adjustments.pdf --[document: acquisition agreement]--> 01_acquisition_agreement.pdf + 05_financial_adjustments.pdf --[document: closing checklist]--> 10_closing_checklist.pdf + 05_financial_adjustments.pdf --[document: due diligence report]--> 02_due_diligence_report.pdf + 05_financial_adjustments.pdf --[document: risk assessment]--> 04_risk_assessment.pdf + 06_legal_opinion.pdf --[document: acquisition agreement]--> 01_acquisition_agreement.pdf + 06_legal_opinion.pdf --[document: due diligence report]--> 02_due_diligence_report.pdf + 06_legal_opinion.pdf --[document: ip certification]--> 03_ip_certification.pdf + 06_legal_opinion.pdf --[document: regulatory approval]--> 08_regulatory_approval.pdf + 07_nda.pdf --[document: acquisition agreement]--> 01_acquisition_agreement.pdf + 07_nda.pdf --[document: due diligence report]--> 02_due_diligence_report.pdf + 07_nda.pdf --[document: ip certification]--> 03_ip_certification.pdf + 08_regulatory_approval.pdf --[document: acquisition agreement]--> 01_acquisition_agreement.pdf + 08_regulatory_approval.pdf --[document: closing checklist]--> 10_closing_checklist.pdf + 08_regulatory_approval.pdf --[document: due diligence report]--> 02_due_diligence_report.pdf + 08_regulatory_approval.pdf --[document: risk assessment]--> 04_risk_assessment.pdf + 09_customer_consents.pdf --[document: acquisition agreement]--> 01_acquisition_agreement.pdf + 09_customer_consents.pdf --[document: due diligence report]--> 02_due_diligence_report.pdf + 10_closing_checklist.pdf --[document: acquisition agreement]--> 01_acquisition_agreement.pdf + 10_closing_checklist.pdf --[document: due diligence report]--> 02_due_diligence_report.pdf + 10_closing_checklist.pdf --[document: financial adjustments]--> 05_financial_adjustments.pdf + 10_closing_checklist.pdf --[document: regulatory approval]--> 08_regulatory_approval.pdf + +unresolved refs: {('exhibit','exhibit a'): 4, ('exhibit','schedule 3'): 5, + ('exhibit','exhibit b'): 4, ('exhibit','exhibit c'): 6, ('exhibit','schedule 1'): 5, + ('exhibit','schedule 2'): 4, ('section','section 1.5'): 1, ('section','section 2.2'): 2, + ('section','section 4.1'): 1, ('section','section 1.1'): 1, ('section','section 4'): 1} + +self-resolution check: NONE (good) +``` + +**34 resolved references, 9 of 10 documents linked, no self-edges** โ€” exactly what +the gate predicted from corpus text. `07_nda.pdf` is the one document nothing +points *at*; it points at three others. The 34 unresolved refs are the Exhibits +and Schedules, which are sections *inside* `01_acquisition_agreement.pdf` rather +than separate files, plus the bare `section N` forms โ€” correctly left `null`, +unchanged by the fix and not fixable by a filename-based resolver. + +`acq+docs` shows the same 34 acquisition edges plus 59 more from +`Documentation/*.md` title mentions (93 resolved / 21 documents). + +### 1.3 Re-baseline โ€” no drift + +Gold coverage on both rebuilt indexes: **24/24** (`acq`) and **48/48** +(`acq+docs`), `coverage_failures` empty. Chunk counts unchanged (13 / 373). + +| corpus | slice | n | R@5 | R@10 | R@20 | nDCG@10 (1st) | previous retry-off figure | +|---|---|---|---|---|---|---|---| +| `acq` | all | 24 | 0.958 | 1.000 | 1.000 | **0.8101** | 0.8101 โœ… | +| `acq` | `requires_crossref=true` | 11 | 1.000 | 1.000 | 1.000 | **0.7477** | 0.7477 โœ… | +| `acq` | control (`=false`) | 13 | 0.923 | 1.000 | 1.000 | **0.8628** | 0.8628 โœ… | +| `acq+docs` | all | 48 | 0.854 | 0.896 | 0.958 | **0.7194** | 0.7194 โœ… | +| `acq+docs` | `requires_crossref=true` | 11 | 1.000 | 1.000 | 1.000 | **0.7477** | 0.7477 โœ… | +| `acq+docs` | control (`=false`) | 13 | 0.769 | 0.769 | 0.846 | **0.7731** | 0.7731 โœ… | + +**Zero drift to four decimals on every cell.** The crossref-slice first-stage +nDCG@10 baseline of 0.748 reproduces exactly. That is the expected result and it +is worth stating why: cross-reference extraction writes only chunk *metadata* โ€” +it moves neither the `text` column nor a bit of any vector โ€” so a rebuild that +adds 34 resolved references cannot change a first-stage ranking. It is also the +control that says the rebuild introduced nothing else. + +The `final == first_stage` invariant passed on both rebuild runs (24/24 and +48/48 queries, chunk-id order and both metric families). + +--- + +## 2. Item 4.2 โ€” the cross-reference hop A/B + +Twelve runs: `{acq, acq+docs}` ร— `k โˆˆ {3, 5, 20}` ร— `{hop off, hop on}`, all +`--retry off`, all on the rebuilt indexes. Command shape: + +``` +.venv/bin/python eval/run_eval.py --corpus <acq|acq+docs> --retry off \ + --crossref-hop <off|on> --overview-prefilter off --k <3|5|20> \ + --json-out eval/results/phase4_w3_42_<tag>_k<k>_hop_<arm>.json +``` + +### 2.1 The full matrix + +`hop q` = queries on which the hop fired. `chunks` = chunks it appended in +total. `hit_src` = queries where a hopped chunk came from a document listed in +the gold row's `expected_sources` (document-level hop precision). `rel` = +queries where a hopped chunk actually contains the gold `expected` text +(text-level hop precision). + +| corpus | k | arm | slice | n | R@5 | R@10 | R@20 | nDCG@10 (1st) | R@5 (fin) | R@10 (fin) | R@20 (fin) | **nDCG@10 (fin)** | hop q | chunks | hit_src | rel | +|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---| +| acq | 20 | off | all | 24 | 0.958 | 1.000 | 1.000 | 0.8101 | 0.958 | 1.000 | 1.000 | **0.8101** | 0 | 0 | 0 | 0 | +| acq | 20 | **on** | all | 24 | 0.958 | 1.000 | 1.000 | 0.8101 | 0.958 | 1.000 | 1.000 | **0.8101** | **0** | 0 | 0 | 0 | +| acq | 20 | off | xref | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | 0 | 0 | 0 | 0 | +| acq | 20 | **on** | xref | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | **0** | 0 | 0 | 0 | +| acq | 20 | off | control | 13 | 0.923 | 1.000 | 1.000 | 0.8628 | 0.923 | 1.000 | 1.000 | **0.8628** | 0 | 0 | 0 | 0 | +| acq | 20 | **on** | control | 13 | 0.923 | 1.000 | 1.000 | 0.8628 | 0.923 | 1.000 | 1.000 | **0.8628** | **0** | 0 | 0 | 0 | +| acq | 5 | off | all | 24 | 0.917 | 0.917 | 0.917 | 0.7875 | 0.917 | 0.917 | 0.917 | **0.7875** | 0 | 0 | 0 | 0 | +| acq | 5 | **on** | all | 24 | 0.917 | 0.917 | 0.917 | 0.7875 | 0.917 | **0.958** | **0.958** | **0.8024** | **24** | 34 | 1 | 1 | +| acq | 5 | off | xref | 11 | 1.000 | 1.000 | 1.000 | 0.7358 | 1.000 | 1.000 | 1.000 | **0.7358** | 0 | 0 | 0 | 0 | +| acq | 5 | **on** | xref | 11 | 1.000 | 1.000 | 1.000 | 0.7358 | 1.000 | 1.000 | 1.000 | **0.7358** | **11** | 16 | **0** | **0** | +| acq | 5 | off | control | 13 | 0.846 | 0.846 | 0.846 | 0.8313 | 0.846 | 0.846 | 0.846 | **0.8313** | 0 | 0 | 0 | 0 | +| acq | 5 | **on** | control | 13 | 0.846 | 0.846 | 0.846 | 0.8313 | 0.846 | **0.923** | **0.923** | **0.8587** | **13** | 18 | 1 | 1 | +| acq | 3 | off | all | 24 | 0.792 | 0.792 | 0.792 | 0.8036 | 0.792 | 0.792 | 0.792 | **0.8036** | 0 | 0 | 0 | 0 | +| acq | 3 | **on** | all | 24 | 0.792 | 0.792 | 0.792 | 0.8036 | **0.875** | **0.875** | **0.875** | **0.8394** | **24** | 37 | 2 | 2 | +| acq | 3 | off | xref | 11 | 0.909 | 0.909 | 0.909 | 0.7868 | 0.909 | 0.909 | 0.909 | **0.7868** | 0 | 0 | 0 | 0 | +| acq | 3 | **on** | xref | 11 | 0.909 | 0.909 | 0.909 | 0.7868 | 0.909 | 0.909 | 0.909 | **0.7868** | **11** | 18 | **0** | **0** | +| acq | 3 | off | control | 13 | 0.692 | 0.692 | 0.692 | 0.8178 | 0.692 | 0.692 | 0.692 | **0.8178** | 0 | 0 | 0 | 0 | +| acq | 3 | **on** | control | 13 | 0.692 | 0.692 | 0.692 | 0.8178 | **0.846** | **0.846** | **0.846** | **0.8840** | **13** | 19 | 2 | 2 | +| acq+docs | 20 | off | all | 48 | 0.854 | 0.896 | 0.958 | 0.7194 | 0.854 | 0.896 | 0.958 | **0.7194** | 0 | 0 | 0 | 0 | +| acq+docs | 20 | **on** | all | 48 | 0.854 | 0.896 | 0.958 | 0.7194 | 0.854 | 0.896 | 0.958 | **0.7194** | **14** | 30 | **0** | **0** | +| acq+docs | 20 | off | xref | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | 0 | 0 | 0 | 0 | +| acq+docs | 20 | **on** | xref | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | **0** | 0 | 0 | 0 | +| acq+docs | 20 | off | control | 13 | 0.769 | 0.769 | 0.846 | 0.7731 | 0.769 | 0.769 | 0.846 | **0.7731** | 0 | 0 | 0 | 0 | +| acq+docs | 20 | **on** | control | 13 | 0.769 | 0.769 | 0.846 | 0.7731 | 0.769 | 0.769 | 0.846 | **0.7731** | **8** | 12 | **0** | **0** | +| acq+docs | 5 | off | all | 48 | 0.854 | 0.854 | 0.854 | 0.6996 | 0.854 | 0.854 | 0.854 | **0.6996** | 0 | 0 | 0 | 0 | +| acq+docs | 5 | **on** | all | 48 | 0.854 | 0.854 | 0.854 | 0.6996 | 0.854 | **0.917** | **0.917** | **0.7104** | **33** | 60 | 1 | 3 | +| acq+docs | 5 | off | xref | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | 0 | 0 | 0 | 0 | +| acq+docs | 5 | **on** | xref | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | **0.7477** | **11** | 13 | **0** | **0** | +| acq+docs | 5 | off | control | 13 | 0.769 | 0.769 | 0.769 | 0.7577 | 0.769 | 0.769 | 0.769 | **0.7577** | 0 | 0 | 0 | 0 | +| acq+docs | 5 | **on** | control | 13 | 0.769 | 0.769 | 0.769 | 0.7577 | 0.769 | **0.846** | **0.846** | **0.7851** | **12** | 17 | 1 | 1 | +| acq+docs | 3 | off | all | 48 | 0.708 | 0.708 | 0.708 | 0.6446 | 0.708 | 0.708 | 0.708 | **0.6446** | 0 | 0 | 0 | 0 | +| acq+docs | 3 | **on** | all | 48 | 0.708 | 0.708 | 0.708 | 0.6446 | **0.771** | **0.792** | **0.792** | **0.6671** | **33** | 64 | 1 | 4 | +| acq+docs | 3 | off | xref | 11 | 0.909 | 0.909 | 0.909 | 0.7413 | 0.909 | 0.909 | 0.909 | **0.7413** | 0 | 0 | 0 | 0 | +| acq+docs | 3 | **on** | xref | 11 | 0.909 | 0.909 | 0.909 | 0.7413 | 0.909 | 0.909 | 0.909 | **0.7413** | **11** | 16 | **0** | **0** | +| acq+docs | 3 | off | control | 13 | 0.692 | 0.692 | 0.692 | 0.7408 | 0.692 | 0.692 | 0.692 | **0.7408** | 0 | 0 | 0 | 0 | +| acq+docs | 3 | **on** | control | 13 | 0.692 | 0.692 | 0.692 | 0.7408 | **0.769** | **0.769** | **0.769** | **0.7740** | **12** | 18 | 1 | 1 | + +### 2.2 First stage is identical across every arm + +Checked per query, not in aggregate โ€” `recall@{5,10,20}`, `ndcg10_first_stage` +and the candidate count, hop-off vs hop-on: + +``` +acq k=20: first-stage recall+nDCG+candidate-count identical on 24/24 queries: True +acq k=5 : first-stage recall+nDCG+candidate-count identical on 24/24 queries: True +acq k=3 : first-stage recall+nDCG+candidate-count identical on 24/24 queries: True +acq+docs k=20: first-stage recall+nDCG+candidate-count identical on 48/48 queries: True +acq+docs k=5 : first-stage recall+nDCG+candidate-count identical on 48/48 queries: True +acq+docs k=3 : first-stage recall+nDCG+candidate-count identical on 48/48 queries: True +``` + +That is the contract (`_crossref_hop` only ever appends to `documents`) verified +on 216 query pairs, and it means every difference in the *final* columns is the +hop and nothing else. + +### 2.3 What actually happened + +**The hop now fires โ€” a lot.** 0 โ†’ 24/24 queries on `acq` at k=3 and k=5, and +33/48 on `acq+docs`. The resolver fix is what changed; nothing else did. + +**At k=20 on `acq` it still fires zero times, and that is correct.** `acq` is 13 +chunks; k=20 sweeps the entire corpus, so all ten documents are already +`represented` and the "not already a candidate" guard suppresses every target. +There is nothing to fetch. This is the structural constraint +`phase4-crossref-prefilter.md` names, now confirmed on an index that *has* +resolvable references: the hop is a candidate-selection mechanism and cannot act +on a corpus smaller than the candidate budget. + +**At k=20 on `acq+docs` it fires 14 times and moves nothing** (0.7194 โ†’ 0.7194, +recall bit-identical). Hopped chunks are appended, so at k=20 they land at ranks +21+ and cannot enter nDCG@10 by construction โ€” and recall@20 is likewise a +prefix that ends before them. `hit_expected_source = 0` on all 14. + +**At small k it moves final recall โ€” but not on the slice it was built for.** +Every gain is in the `requires_crossref=false` control slice: + +| corpus | k | slice | R@10 (fin) off โ†’ on | nDCG@10 (fin) off โ†’ on | +|---|---|---|---|---| +| acq | 5 | **xref (n=11)** | 1.000 โ†’ 1.000 (**+0.000**) | 0.7358 โ†’ 0.7358 (**+0.0000**) | +| acq | 5 | control (n=13) | 0.846 โ†’ 0.923 (+0.077) | 0.8313 โ†’ 0.8587 (+0.0274) | +| acq | 3 | **xref (n=11)** | 0.909 โ†’ 0.909 (**+0.000**) | 0.7868 โ†’ 0.7868 (**+0.0000**) | +| acq | 3 | control (n=13) | 0.692 โ†’ 0.846 (+0.154) | 0.8178 โ†’ 0.8840 (+0.0662) | +| acq+docs | 5 | **xref (n=11)** | 1.000 โ†’ 1.000 (**+0.000**) | 0.7477 โ†’ 0.7477 (**+0.0000**) | +| acq+docs | 5 | control (n=13) | 0.769 โ†’ 0.846 (+0.077) | 0.7577 โ†’ 0.7851 (+0.0274) | +| acq+docs | 3 | **xref (n=11)** | 0.909 โ†’ 0.909 (**+0.000**) | 0.7413 โ†’ 0.7413 (**+0.0000**) | +| acq+docs | 3 | control (n=13) | 0.692 โ†’ 0.769 (+0.077) | 0.7408 โ†’ 0.7740 (+0.0332) | + +**On the `requires_crossref` slice the hop changes nothing at any k on either +corpus, and `hit_expected_source` is 0/11 in every single cell.** Not one hop on +a cross-reference query landed in a gold source document. + +Two reasons, both true at once: + +1. The slice is at ceiling before the hop runs (recall 0.909โ€“1.000 at every k โ€” + this is the finding `BASELINE.md` already records: "the crossref slice is + **not** weak"). A mechanism that only *adds* candidates cannot help a query + whose answer is already retrieved. +2. The hop picks the wrong target. See ยง2.5. + +**The hop never hurt anything.** Recall never fell, nDCG@10 (final) never fell, +in any of the 36 rows above. Appending cannot demote. + +### 2.4 The honest control: is the hop better than just retrieving more chunks? + +The hop's gains all come from making the final list longer. The fair comparison +is therefore not off-vs-on at equal `k`, but off-vs-on at **equal final list +length**. Mean final list size and final recall/nDCG: + +``` +acq k=3 hop ON : mean final list = 4.54 chunks | R@10f=0.875 nDCG@10f=0.8394 +acq k=5 hop ON : mean final list = 6.42 chunks | R@10f=0.958 nDCG@10f=0.8024 +acq k=3 hop off: mean final list = 3.00 chunks | R@10f=0.792 nDCG@10f=0.8036 +acq k=4 hop off: mean final list = 4.00 chunks | R@10f=0.875 nDCG@10f=0.7802 +acq k=5 hop off: mean final list = 5.00 chunks | R@10f=0.917 nDCG@10f=0.7875 +acq k=6 hop off: mean final list = 6.00 chunks | R@10f=0.958 nDCG@10f=0.8115 + +acq+docs k=3 hop ON : mean final list = 4.33 chunks | R@10f=0.792 nDCG@10f=0.6671 +acq+docs k=5 hop ON : mean final list = 6.25 chunks | R@10f=0.917 nDCG@10f=0.7104 +acq+docs k=3 hop off: mean final list = 3.00 chunks | R@10f=0.708 nDCG@10f=0.6446 +acq+docs k=4 hop off: mean final list = 4.00 chunks | R@10f=0.812 nDCG@10f=0.6858 +acq+docs k=5 hop off: mean final list = 5.00 chunks | R@10f=0.854 nDCG@10f=0.6996 +acq+docs k=6 hop off: mean final list = 6.00 chunks | R@10f=0.875 nDCG@10f=0.7103 +``` + +Budget-matched, four cells: + +| corpus | hop arm | list size | R@10 (fin) | nearest hop-off arm | list size | R@10 (fin) | verdict | +|---|---|---|---|---|---|---|---| +| acq | k=3 on | 4.54 | 0.875 | k=4 off | 4.00 | 0.875 | **tie, hop costs 0.54 more chunks** | +| acq | k=5 on | 6.42 | 0.958 | k=6 off | 6.00 | 0.958 | **tie, hop costs 0.42 more chunks** (and nDCG 0.8024 vs 0.8115 โ€” hop slightly worse) | +| acq+docs | k=3 on | 4.33 | 0.792 | k=4 off | 4.00 | 0.812 | **plain k=4 wins**, with fewer chunks | +| acq+docs | k=5 on | 6.25 | 0.917 | k=6 off | 6.00 | 0.875 | **hop wins**, +0.042 recall for +0.25 chunks | + +One cell out of four where the hop beats simply raising `k`. That is the single +most important number in this document: on this corpus and this gold set, **the +cross-reference hop is, to a first approximation, an expensive way of retrieving +one more chunk.** It costs an extra filtered vector search per query; raising +`k` costs nothing. + +### 2.5 Mechanism finding โ€” the hop reliably picks the hub, not the target + +Per-query hop targets, `acq`, k=3, hop on (`hit_src` / `rel` are the precision +flags; `xref` is `requires_crossref`): + +``` +id xref anchor -> hop target || expected_sources +acq_q01 False 05_financial_adjustments.pdf --[risk assessment]--> 04_risk_assessment.pdf || ['01_acquisition_agreement.pdf'] +acq_q02 False 07_nda.pdf --[due diligence report]->02_due_diligence_report.pdf|| ['07_nda.pdf'] +acq_q03 False 03_ip_certification.pdf --[risk assessment]--> 04_risk_assessment.pdf || ['03_ip_certification.pdf'] +acq_q04 False 06_legal_opinion.pdf --[due diligence report]->02_due_diligence_report.pdf|| ['02_due_diligence_report.pdf'] hit_src=True rel=True +acq_q05 False 07_nda.pdf --[due diligence report]->02_due_diligence_report.pdf|| ['07_nda.pdf', '07_nda.pdf'] +acq_q06 False 07_nda.pdf --[due diligence report]->02_due_diligence_report.pdf|| ['07_nda.pdf'] +acq_q07 False 08_regulatory_approval.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['08_regulatory_approval.pdf'] +acq_q08 False 06_legal_opinion.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['06_legal_opinion.pdf'] +acq_q09 False 05_financial_adjustments.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['02_due_diligence_report.pdf', '05_financial_adjustments.pdf'] +acq_q10 False 09_customer_consents.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['04_risk_assessment.pdf', '09_customer_consents.pdf'] +acq_q11 False 05_financial_adjustments.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['05_financial_adjustments.pdf', '10_closing_checklist.pdf'] +acq_q12 False 05_financial_adjustments.pdf --[due diligence report]->02_due_diligence_report.pdf|| ['02_due_diligence_report.pdf'] hit_src=True rel=True +acq_q13 True 01_acquisition_agreement.pdf --[due diligence report]->02_due_diligence_report.pdf|| ['08_regulatory_approval.pdf'] +acq_q14 True 04_risk_assessment.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['04_risk_assessment.pdf'] +acq_q15 True 05_financial_adjustments.pdf --[due diligence report]->02_due_diligence_report.pdf|| ['05_financial_adjustments.pdf'] +acq_q16 True 03_ip_certification.pdf --[risk assessment]--> 04_risk_assessment.pdf || ['03_ip_certification.pdf'] +acq_q17 True 07_nda.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['02_due_diligence_report.pdf'] +acq_q18 True 04_risk_assessment.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['03_ip_certification.pdf'] +acq_q19 True 06_legal_opinion.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['09_customer_consents.pdf'] ร—3 +acq_q20 True 08_regulatory_approval.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['08_regulatory_approval.pdf'] +acq_q21 True 10_closing_checklist.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['05_financial_adjustments.pdf'] +acq_q22 True 08_regulatory_approval.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['10_closing_checklist.pdf'] +acq_q23 True 01_acquisition_agreement.pdf --[due diligence report]->02_due_diligence_report.pdf|| ['05_financial_adjustments.pdf', '08_regulatory_approval.pdf'] +acq_q24 False 08_regulatory_approval.pdf --[acquisition agreement]>01_acquisition_agreement.pdf|| ['08_regulatory_approval.pdf'] +``` + +Target concentration across the on-arms: + +``` +acq k=3 (24 targets): 13 ร— 01_acquisition_agreement.pdf, 8 ร— 02_due_diligence_report.pdf, 3 ร— 04_risk_assessment.pdf +acq k=5 (24 targets): 10 ร— 01_acquisition_agreement.pdf, 8 ร— 02_due_diligence_report.pdf, 3 ร— 08_regulatory_approval.pdf, 1 each ร— 3 others +acq+docs k=3 (33 targets): 10 ร— 01_acquisition_agreement.pdf, 9 ร— 02_due_diligence_report.pdf, 9 ร— retrieval_pipeline.md, 2 ร— 04_risk_assessment.pdf, 1 each ร— 3 others +acq+docs k=5 (33 targets): 10 ร— 02_due_diligence_report.pdf, 7 ร— retrieval_pipeline.md, 6 ร— 01_acquisition_agreement.pdf, 3 ร— verifier.md, โ€ฆ +``` + +**21 of 24 hops on `acq` k=3 go to one of two documents.** The reason is in +`rag_system/pipelines/retrieval_pipeline.py::_crossref_hop`: candidate targets +are collected by scanning the top-3 candidates in rank order and, within each, +the chunk's `crossrefs` list *in order of first appearance in the text*; the list +is then truncated with `targets = targets[:max_hops]` and `max_hops` is 1. So the +winner is whichever document the top candidate happens to mention first โ€” and in +a deal room every document opens by naming the master agreement. Query relevance +never enters the target choice at any point. + +**This is a design finding, not a crash, and it is deliberately not fixed here** +(`rag_system/` is outside this wave's ownership). The proposed diff, for the +gate: + +```python +# rag_system/pipelines/retrieval_pipeline.py, in _crossref_hop, replacing +# targets = targets[:max_hops] + + # Order candidate targets by evidence that the query is about them, + # not by where the reference happens to sit in the chunk text. Two + # signals, both free: how many of the top-3 candidates point at the + # same document (agreement), and how highly ranked the referring + # candidate was. Without this, `max_hops=1` on a corpus with a hub + # document sends every query to the hub โ€” measured 21/24 on `acq`. + votes: Dict[str, int] = {} + for t in targets: + votes[t["target_doc"]] = votes.get(t["target_doc"], 0) + 1 + targets.sort(key=lambda t: (-votes[t["target_doc"]], t["from_rank"])) + targets = targets[:max_hops] +``` + +Honest caveat on that diff: it is **untested and unmeasured**, it would not have +rescued a single `requires_crossref` row here (those rows are already at recall +ceiling, so there is nothing for a better-chosen hop to add), and on this corpus +the hub is also the most-voted-for document, so it may well change nothing. It is +recorded because the current selection rule is indefensible on inspection, not +because there is evidence it costs recall. + +A second, cheaper option the gate may prefer: raise `max_hops` from 1 to 2โ€“3, so +the hub does not crowd out the specific reference. That trades precision for +context length and needs its own arm. + +--- + +## 3. Item 4.3 โ€” the overview prefilter + +### 3.1 Making it measurable: one additive change to `eval/run_eval.py` + +Before this wave, 4.3 could not be measured at all. The prefilter scores the +query against `index_store/overviews/<id>.vectors.npz`, a sidecar written at the +end of an index build by `IndexingPipeline` step 4 โ€” and only when +`overview.enabled` is true. The eval harness hard-set `cfg["overview"] = +{"enabled": False}` (one LLM call per document, same reason enrichment is off), +so no eval index has ever had a sidecar, and `RetrievalPipeline._overview_vectors` +correctly degraded to a one-line "no embedded overviews were found" no-op. + +The change, all inside `eval/run_eval.py`, all additive, default unchanged: + +* **`--overviews {off,on}`**, default `off`. `on` sets + `cfg["overview"] = {"enabled": True, "embed": True}`. +* **`cfg["overview_path"]` is redirected into the corpus's own eval index + directory** (`eval/.eval_indexes/<embedder>/<corpus>_ov/overviews.jsonl`) + rather than the repo's shared `index_store/overviews/`. The sidecar is then + owned by the index, deleted with it, and cannot be read by the wrong corpus. + `RetrievalPipeline._overview_vectors` resolution rule 2 (`config["overview_path"]` + with `.jsonl` โ†’ `.vectors.npz`) picks it up with no pipeline change. +* **An overview build gets its own index directory** (`<corpus>_ov`). Sharing one + would make every alternation between `--overviews off` and `--overviews on` a + full re-index, and would delete an index another process is reading. +* **`index_fingerprint` gains `"overviews": True` โ€” only when true.** Adding the + key unconditionally would have invalidated every index cached before the flag + existed and forced needless rebuilds of `mixed`, `docs`, `atlas7`, `hr`. +* The run header and `run.overviews` in the results JSON report the state. + +Nothing else changed; `py_compile` clean. + +**Verification that this does not alter what is being measured.** An +overview-enabled build must be chunk-for-chunk identical to a normal one โ€” +`OverviewBuilder.build_and_store` only appends a JSONL line. Checked against the +`acq_plus_docs` index built in ยง1: + +``` +chunks: 373 373 +chunk_id sets identical: True +text identical : True +vector shapes : (373, 1024) (373, 1024) +max abs vector delta : 0.0 +``` + +And the sidecar really is produced, in the right place: + +``` +๐Ÿ”— Cross-references: 215 reference(s) in 70 chunk(s); 93 resolved to 21 document(s). +๐Ÿงญ Embedded 23 document overview(s) โ†’ /โ€ฆ/eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq_plus_docs_ov/overviews.vectors.npz + +$ ls eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq_plus_docs_ov/ +eval_acq_plus_docs.built.json eval_acq_plus_docs.lance overviews.jsonl overviews.vectors.npz +``` + +`index_store/overviews/` was untouched (its three files all predate this wave). + +### 3.2 The arms + +``` +for m in off boost restrict; do + .venv/bin/python eval/run_eval.py --corpus acq+docs --overviews on --retry off \ + --crossref-hop off --overview-prefilter $m \ + --json-out eval/results/phase4_w3_43_acqdocs_ov_$m.json + .venv/bin/python eval/run_eval.py --corpus mixed --overviews on --retry off \ + --crossref-hop off --overview-prefilter $m \ + --json-out eval/results/phase4_w3_43_mixed_ov_$m.json +done +``` + +`top_documents = 5` (the profile default), k = 20, 23 overviews on `acq+docs`, +15 on `mixed`. The prefilter loaded and fired on every query โ€” verbatim, from +the `mixed` run log: + +``` +๐Ÿงญ Overview prefilter: 15 document overview(s) loaded from /โ€ฆ/mixed_ov/overviews.vectors.npz. +๐Ÿงญ Overview prefilter selected 5 document(s): atlas7_service_manual.pdf, design_rationale.md, quick_start.md, northwind_leave_policy.pdf, prompt_inventory.md +๐Ÿงญ Overview prefilter selected 5 document(s): atlas7_service_manual.pdf, quick_start.md, architecture_overview.md, prompt_inventory.md, northwind_leave_policy.pdf +โ€ฆ +$ grep -c "Overview prefilter selected" <log> +72 +``` + +### 3.3 Results + +| corpus | arm | slice | n | R@5 | R@10 | R@20 | nDCG@10 (1st) | R@5 (fin) | R@10 (fin) | R@20 (fin) | nDCG@10 (fin) | +|---|---|---|---|---|---|---|---|---|---|---|---| +| acq+docs | off | all | 48 | 0.854 | 0.896 | 0.958 | **0.7194** | 0.854 | 0.896 | 0.958 | 0.7194 | +| acq+docs | **boost** | all | 48 | **0.812** | **0.917** | 0.958 | **0.7017** | 0.812 | 0.917 | 0.958 | 0.7017 | +| acq+docs | **restrict** | all | 48 | **0.812** | **0.854** | **0.896** | **0.6951** | 0.812 | 0.854 | 0.896 | 0.6951 | +| acq+docs | off | `requires_crossref=true` | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | 0.7477 | +| acq+docs | boost | `requires_crossref=true` | 11 | 1.000 | 1.000 | 1.000 | **0.7261** | 1.000 | 1.000 | 1.000 | 0.7261 | +| acq+docs | restrict | `requires_crossref=true` | 11 | 1.000 | 1.000 | 1.000 | 0.7477 | 1.000 | 1.000 | 1.000 | 0.7477 | +| acq+docs | off | control (`=false`) | 13 | 0.769 | 0.769 | 0.846 | 0.7731 | 0.769 | 0.769 | 0.846 | 0.7731 | +| acq+docs | boost | control (`=false`) | 13 | **0.846** | **0.846** | 0.846 | **0.8790** | 0.846 | 0.846 | 0.846 | 0.8790 | +| acq+docs | restrict | control (`=false`) | 13 | **0.846** | **0.846** | **0.923** | **0.8480** | 0.846 | 0.846 | 0.923 | 0.8480 | +| acq+docs | off | **multi_document=true** | 4 | 0.750 | 0.750 | 0.750 | 0.7803 | 0.750 | 0.750 | 0.750 | 0.7803 | +| acq+docs | **boost** | **multi_document=true** | 4 | 0.750 | 0.750 | 0.750 | **0.8926** | 0.750 | 0.750 | 0.750 | 0.8926 | +| acq+docs | **restrict** | **multi_document=true** | 4 | 0.750 | 0.750 | **1.000** | **0.7394** | 0.750 | 0.750 | 1.000 | 0.7394 | +| mixed | off | all | 72 | 0.944 | 0.972 | 1.000 | **0.8873** | 0.944 | 0.972 | 1.000 | 0.8873 | +| mixed | **boost** | all | 72 | **0.889** | 0.972 | 1.000 | **0.8662** | 0.889 | 0.972 | 1.000 | 0.8662 | +| mixed | **restrict** | all | 72 | **0.889** | **0.931** | **0.944** | **0.8740** | 0.889 | 0.931 | 0.944 | 0.8740 | + +`mixed` has no `multi_document` rows (the key lives only on the `acq` gold set), +so that slice is reported for `acq+docs` only. `multi_document` is a **top-level +gold-row key, not a `dimensions` key**, so `run_eval.py`'s `by_dimension` does +not surface it; the four rows above were sliced by joining the results JSON back +to `eval/goldset/acquisition.jsonl`. + +The `final == first_stage` invariant passed on all six arms (hop off, rerank +off), so every 4.3 number above is a pure first-stage effect. + +### 3.4 The harm check โ€” `restrict` hides the answer document + +`recall@20` is where a document the prefilter excluded shows up as a total miss. +Per-query transitions against the `off` arm: + +``` +acq+docs boost : lost@20 [] gained@20 [] +acq+docs restrict: lost@20 [docs_d04, docs_d08, docs_d12, docs_d13] gained@20 [acq_q09] +mixed boost : lost@20 [] gained@20 [] +mixed restrict: lost@20 [docs_d12, docs_d13, docs_d20, docs_d24] gained@20 [] +``` + +**`restrict` costs four queries their answer entirely on each corpus** โ€” the +answer-bearing chunk is not merely demoted, it is unreachable. `boost` loses +nothing at @20 on either corpus, exactly as designed (it reorders, it never +drops). + +A worked example, `docs_d12` *"How do I change the embedding model?"*. The gold +string `"Changing the embedding model requires re-indexing"` lives in +`deployment_guide.md`, `retrieval_pipeline.md` and `system_overview.md`. The +prefilter's top-5 for that query was: + +``` +๐Ÿงญ Overview prefilter selected 5 document(s): quick_start.md, indexing_pipeline.md, + installation_guide.md, triage_system.md, architecture_overview.md +``` + +None of the three documents that contain the answer. Under `boost` those +documents merely get no second RRF leg and the query still scores; under +`restrict` the LanceDB `document_id IN (โ€ฆ)` prefilter removes them and recall +goes 1 โ†’ 0. The overviews are not wrong โ€” a question about *changing a model* +does read like a setup question โ€” they are just a 120-token summary standing in +for a 30-chunk document. + +At @5 both modes cost the same four documentation queries and rescue two +(`acq+docs`: lost `docs_d04, docs_d08, docs_d10, docs_d12`, gained `acq_q04, +docs_d17`; `mixed`: lost `docs_d10, docs_d12, docs_d20, docs_d24`, gained none). + +### 3.5 Reading the 4.3 result + +`boost` is **strongly corpus-dependent, and the split is legible**: + +* On `acq+docs` it is a real gain where the corpus is *heterogeneous* โ€” the + `acq` control slice goes 0.769 โ†’ 0.846 recall@10 and **0.7731 โ†’ 0.8790 + nDCG@10 (+0.106)**, and the `multi_document` slice **0.7803 โ†’ 0.8926 + (+0.112)**. When ten M&A PDFs sit in a table with thirteen localGPT + documentation files, a document-level prior is genuinely informative. +* On `mixed` โ€” atlas7 + hr + docs, where twelve of the fifteen documents are + localGPT documentation whose overviews all say roughly "a technical document + about the localGPT RAG system" โ€” it is a **loss**: recall@5 0.944 โ†’ 0.889 and + nDCG@10 0.8873 โ†’ 0.8662. The prior is uninformative and the RRF second leg is + just noise promoting the wrong documentation file. +* On the whole `acq+docs` set the two effects cancel into a small net loss + (nDCG@10 0.7194 โ†’ 0.7017, recall@5 0.854 โ†’ 0.812, recall@10 0.896 โ†’ 0.917). + +`restrict` is worse than `boost` on every aggregate on both corpora **and** it is +the only arm that destroys recall@20. Its one apparent win โ€” `multi_document` +recall@20 0.750 โ†’ 1.000 (n=4, i.e. one query) โ€” is not worth the four queries it +kills on the same run. + +The `requires_crossref` slice is untouched by `restrict` (0.7477 โ†’ 0.7477) and +mildly hurt by `boost` (0.7477 โ†’ 0.7261). 4.3 was never aimed at that slice. + +--- + +## 4. Caveats, stated plainly + +* **n = 11** on the `requires_crossref` slice and **n = 4** on + `multi_document`. One query is 0.09 and 0.25 of those figures respectively. + The `multi_document` "0.750 โ†’ 1.000" in ยง3.3 is literally one query changing. + Neither slice can support a fine-grained call. +* **`acq` is 13 chunks.** At the shipped `k = 20` no candidate-selection change + can move a metric on it, which is why the 4.2 evidence lives at k=3 and k=5 โ€” + values the product does not use. The k=20 rows are the ones that describe + shipped behaviour, and there the hop is inert. +* **Retrieval metrics only, no judge.** Nothing here says whether an answer + improved. A hop that puts the right document into the context without changing + its rank is invisible to this harness โ€” the "which document gets cited" gap + `BASELINE.md` names. For 4.2 in particular, the *product* question is whether + the hopped chunk gets cited, and it is unanswered. +* **The `requires_crossref` slice is at recall ceiling before either mechanism + runs.** This is the deepest limitation: the gold set was built to *expose* the + cross-reference failure and does not reproduce it (BASELINE.md ยง "The crossref + slice is **not** weak" gives three reasons). A mechanism that adds candidates + cannot be shown to help queries that already retrieve their answer. **4.2 + remains under-tested rather than disproven.** +* **Overview text is LLM-generated and is not reproducible across rebuilds.** + Within this wave all three arms of each 4.3 comparison read the *same* sidecar, + so the comparison is exact; a future rebuild will produce different overviews + and could move the 4.3 numbers without any code change. The `_ov` index + directories should be preserved, not regenerated, if these numbers are to be + compared against. +* **`acq+docs` full-set figures are lower than `BASELINE.md`'s Phase 4 baseline + row** (0.854 / 0.896 / 0.958 vs 0.917 / 0.958 / 1.000). That is `--retry off` + versus the baseline row's `--retry profile`, not a regression โ€” same caveat as + `phase4-eval-final-metric.md` ยง5. +* **No latency claim.** A concurrent agent shared the Ollama instance. + +--- + +## 5. PROPOSED calls โ€” for the gate, which decides + +### 4.2 cross-reference hop โ€” **PROPOSE: HOLD** (do not adopt, do not reject) + +| evidence | | +|---|---| +| Resolver fix works | 0 โ†’ 34 resolved refs on `acq`, 9/10 documents linked, no self-edges (ยง1.2) | +| Hop fires now | 0 โ†’ 24/24 queries at k=3/k=5 on `acq`, 33/48 on `acq+docs` (ยง2.1) | +| At the shipped k=20 | inert on `acq` (0 hops โ€” corpus smaller than the candidate budget) and null on `acq+docs` (14 hops, every metric bit-identical, `hit_expected_source = 0`) (ยง2.1) | +| On the slice it exists for | **0/11 hops hit a gold source document, at every k, on both corpora; not one metric moved** (ยง2.3) | +| Where it does gain | only the `requires_crossref=false` control slice, at k=3/k=5 (ยง2.3) | +| Budget-matched | beats simply raising `k` in **1 of 4** cells; ties or loses in the other three (ยง2.4) | +| Harm | none measured anywhere โ€” it only appends, and recall never fell (ยง2.1) | +| Target selection | 21/24 hops on `acq` go to one of two hub documents, because targets are ordered by text position and `max_hops=1` (ยง2.5) | + +One-line justification: **the mechanism is now demonstrably alive and +demonstrably harmless, but it produced no gain on the slice it was built for and +does not beat raising `k` at equal context budget โ€” so there is no evidence for +turning it on, and no evidence for tearing it out.** Keep +`indexing.extract_crossrefs` on (free, metadata-only, verified inert on the +chunk index) and keep `retrieval.crossref_hop.enabled = False`. What would +settle it: a judged end-to-end arm on the 11 `requires_crossref` rows, hop on vs +off, measuring *which document gets cited* โ€” the retrieval metric is at ceiling +and cannot answer it. Secondarily, a stricter reference-only gold set whose +queries name the pointer and nothing about the target's content. + +### 4.3 overview prefilter, `boost` mode โ€” **PROPOSE: HOLD** + +| evidence | | +|---|---| +| Now measurable | yes, via `--overviews on` (ยง3.1); sidecar produced, chunk index bit-identical | +| Heterogeneous corpus (`acq` slice of `acq+docs`) | control nDCG@10 **0.7731 โ†’ 0.8790 (+0.106)**, recall@10 0.769 โ†’ 0.846; `multi_document` nDCG@10 **0.7803 โ†’ 0.8926 (+0.112)** | +| Homogeneous corpus (`mixed`) | recall@5 **0.944 โ†’ 0.889**, nDCG@10 **0.8873 โ†’ 0.8662** โ€” a loss | +| Whole `acq+docs` set | recall@5 0.854 โ†’ 0.812, nDCG@10 0.7194 โ†’ 0.7017 โ€” a small net loss | +| Harm at @20 | **none** โ€” nothing lost at recall@20 on either corpus | + +One-line justification: **`boost` helps exactly where document overviews carry +signal (a genuinely mixed collection) and hurts where they do not (fifteen +documents about the same system), and localGPT cannot know which one a user's +index is โ€” so it should not ship on by default on a two-corpus split.** It is the +best candidate here for a per-index opt-in, and the natural next measurement is +`top_documents` sensitivity (5 of 15 documents is a third of `mixed`; 5 of 23 is +a fifth of `acq+docs`) plus a third, deliberately heterogeneous corpus. + +### 4.3 overview prefilter, `restrict` mode โ€” **PROPOSE: REJECT** + +| evidence | | +|---|---| +| `acq+docs` | recall@10 0.896 โ†’ 0.854, recall@20 **0.958 โ†’ 0.896**, nDCG@10 0.7194 โ†’ 0.6951 | +| `mixed` | recall@10 0.972 โ†’ 0.931, recall@20 **1.000 โ†’ 0.944**, nDCG@10 0.8873 โ†’ 0.8740 | +| Harm | **4 queries per corpus lose the answer document entirely** (recall@20 1 โ†’ 0), with a worked trace for `docs_d12` (ยง3.4) | +| Any win? | one query's worth of `multi_document` recall@20, n=4 | + +One-line justification: **it is worse than `boost` on every aggregate on both +corpora and it is the only arm that makes an answer unreachable โ€” a 120-token +LLM summary is not a safe basis for excluding a document from search.** Keep the +code (it costs nothing switched off and `boost` shares its machinery) and never +default it on. + +--- + +## 6. Files touched + +* `eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/{acq,acq_plus_docs}` โ€” + deleted and rebuilt (resolver fix). +* `eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/{acq_plus_docs_ov,mixed_ov}` โ€” + new, overview-enabled builds for the 4.3 arms. Preserve them if the 4.3 + numbers are to be compared against (ยง4, overview reproducibility). +* `eval/results/phase4_w3_*.json` (22 runs) and their `.log` files. +* `eval/run_eval.py` โ€” `--overviews {off,on}` and its plumbing only (ยง3.1). +* `eval/BASELINE.md` โ€” rebuilt-index baselines, dated. +* `eval/decisions/phase4-retrieval-benchmarks.md` โ€” this file. + +Nothing under `rag_system/`, `Documentation/`, `src/` or `backend/`. + +## 7. Reproducing every number above + +```bash +cd /path/to/localGPT + +# 1. rebuild (the resolver fix is index-time) +rm -rf eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq \ + eval/.eval_indexes/microsoft__harrier-oss-v1-0.6b/acq_plus_docs +.venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop off \ + --overview-prefilter off \ + --json-out eval/results/phase4_w3_rebuild_acq_hop_off.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop off \ + --overview-prefilter off \ + --json-out eval/results/phase4_w3_rebuild_acqdocs_hop_off.json + +# 2. the 4.2 matrix (k=20 hop-off arms are the two rebuild runs above) +.venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop on \ + --overview-prefilter off \ + --json-out eval/results/phase4_w3_42_acq_k20_hop_on.json +.venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop on \ + --overview-prefilter off \ + --json-out eval/results/phase4_w3_42_acqdocs_k20_hop_on.json +for k in 5 3; do for arm in off on; do + .venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop $arm \ + --overview-prefilter off --k $k \ + --json-out eval/results/phase4_w3_42_acq_k${k}_hop_${arm}.json + .venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop $arm \ + --overview-prefilter off --k $k \ + --json-out eval/results/phase4_w3_42_acqdocs_k${k}_hop_${arm}.json +done; done + +# 2b. the budget-matched hop-off controls (ยง2.4) +for k in 4 6; do + .venv/bin/python eval/run_eval.py --corpus acq --retry off --crossref-hop off \ + --overview-prefilter off --k $k \ + --json-out eval/results/phase4_w3_42_acq_k${k}_hop_off.json + .venv/bin/python eval/run_eval.py --corpus acq+docs --retry off --crossref-hop off \ + --overview-prefilter off --k $k \ + --json-out eval/results/phase4_w3_42_acqdocs_k${k}_hop_off.json +done + +# 3. the 4.3 arms (first run of each corpus builds the _ov index: 23 / 15 LLM calls) +for m in off boost restrict; do + .venv/bin/python eval/run_eval.py --corpus acq+docs --overviews on --retry off \ + --crossref-hop off --overview-prefilter $m \ + --json-out eval/results/phase4_w3_43_acqdocs_ov_${m}.json + .venv/bin/python eval/run_eval.py --corpus mixed --overviews on --retry off \ + --crossref-hop off --overview-prefilter $m \ + --json-out eval/results/phase4_w3_43_mixed_ov_${m}.json +done +``` + +The crossref verification in ยง1.2, the budget-matched table in ยง2.4, the hop +target listing in ยง2.5 and the `multi_document` / harm slices in ยง3.3โ€“ยง3.4 are +derived from those results JSONs and the built LanceDB tables; they are not +produced by `run_eval.py` itself. diff --git a/eval/decisions/reranker.md b/eval/decisions/reranker.md new file mode 100644 index 00000000..b039c648 --- /dev/null +++ b/eval/decisions/reranker.md @@ -0,0 +1,437 @@ +# Phase 1.1 decision โ€” reranker A/B, measured 2026-08-09 + +> **Superseded as a recommendation, retained as evidence.** The defaults this +> page says it did *not* change were changed at the Phase 1 adoption gate on +> 2026-08-09; what actually shipped, and the joint matrix behind it, is in +> [`../DECISIONS.md`](../DECISIONS.md). Every measurement below stands. + +Roadmap item: [`Documentation/research_roadmap.md`](../../Documentation/research_roadmap.md) ยง1.1. +Baseline this A/B is measured against: [`eval/BASELINE.md`](../BASELINE.md). + +**Every number on this page came from a run executed on this machine on +2026-08-09 (UTC).** Nothing is copied from a leaderboard, estimated or +extrapolated. Failures are reported verbatim. Where a number could not be +measured, it says so. + +Raw outputs (git-ignored, re-run the commands in "Reproducing" to regenerate): +`eval/results/ab_bge_run{1,2}.json`, `eval/results/ab_qwen06_run{1,2}.json`, +`eval/results/ab_qwen4b_run{1,2}.json`. + +**This document changes no default.** `rag_system/main.py` still ships +`BAAI/bge-reranker-v2-m3`. This is a recommendation with its evidence attached. + +--- + +## 1. Integration finding โ€” the `rerankers` library cannot score Qwen3-Reranker + +The roadmap flagged this as an open question ("may need a small custom +scorer"). It was tested, not assumed. + +`rerankers` 0.10.0, the version installed in `.venv`, has **no Qwen3-Reranker +backend**. Its `models/` directory holds ColBERT, FlashRank, LLM-layerwise, +MonoVLM, mxbai-v2, PyLate, RankGPT, RankLLM, T5, UPR and a generic +transformer cross-encoder; the only `qwen` matches in the whole package are +`lightonai/MonoQwen2-VL-v0.1` and a Qwen chat-template string inside +`mxbai_v2.py`. + +Loading `Qwen/Qwen3-Reranker-0.6B` through the shipped path +(`Reranker(model_name, model_type="cross-encoder")` โ€” exactly what +`retrieval_pipeline.py` and `eval/run_eval.py` do) produced, verbatim: + +``` +Some weights of Qwen3ForSequenceClassification were not initialized from the model +checkpoint at Qwen/Qwen3-Reranker-0.6B and are newly initialized: ['score.weight'] +You should probably TRAIN this model on a down-stream task to be able to use it for +predictions and inference. +Loading TransformerRanker model Qwen/Qwen3-Reranker-0.6B (this message can be suppressed by setting verbose=0) +No device set +Using device mps +No dtype set +Using dtype torch.float32 +Loaded model Qwen/Qwen3-Reranker-0.6B +Using device mps. +Using dtype torch.float32. +FAIL cross-encoder ValueError Cannot handle batch sizes > 1 if no padding token is defined. +``` + +Two separate defects in one call: + +1. **Silent correctness failure.** The library builds a + `Qwen3ForSequenceClassification` with a **randomly initialised `score` + head**. Had the batching not thrown, this configuration would have returned + untrained noise while printing "Loaded model" and "AI reranker initialized + successfully" โ€” a wrong-answer failure, not a crash. +2. **Hard failure.** `ValueError: Cannot handle batch sizes > 1 if no padding + token is defined.` โ€” the model's tokenizer has no `pad_token` configured for + the sequence-classification path, so it cannot batch at all. + +Conclusion: a custom scorer was required. It was written. + +### What was implemented + +`QwenRerankerScorer` in +[`rag_system/rerankers/reranker.py`](../../rag_system/rerankers/reranker.py), +implementing the scoring scheme published on the Qwen3-Reranker model card: + +* `AutoModelForCausalLM`, not `AutoModelForSequenceClassification`. +* Tokenizer loaded with `padding_side="left"` (required โ€” the score is read off + the last position of a causal LM). +* Each (query, document) pair wrapped in the model's chat template: + a system turn instructing a yes/no judgment, a user turn + `<Instruct>: โ€ฆ \n<Query>: โ€ฆ \n<Document>: โ€ฆ`, and an assistant turn opened + with an empty `<think></think>` block. +* Score = `softmax` over the `yes`/`no` token logits at the final position, + reported as `P(yes)` in `[0, 1]`. +* fp16 on MPS/CUDA, fp32 on CPU. Batch size 8. Truncation cap 2048 tokens + (padding is to the longest item in the batch, not to the cap). + +Interface: `rank(query, docs, top_k=None) -> [(score, original_index), โ€ฆ]` +sorted descending, plus a dict-in/dict-out `rerank()` mirroring +`CrossEncoderReranker`. Both `RetrievalPipeline.run()` (lines 374โ€“383) and +`eval/run_eval.py`'s `rerank()` already fall back to treating a plain list as +`(score, idx)` pairs when the returned object has no `.results`, so **no change +to either call site was needed** โ€” only to the loader. + +3-pair sanity check on the implemented scorer (executed, not assumed): + +| query | document | 0.6B `P(yes)` | 4B `P(yes)` | +|---|---|---|---| +| "What is the capital of France?" | "Paris is the capital of France." | 0.9974 | 0.9294 | +| " | "The Eiffel Tower is a tower in Paris." | 0.0054 | 0.0633 | +| " | "Bananas are yellow fruit." | 0.0000161 | 0.0000597 | + +### Loader wiring + +`RetrievalPipeline._get_ai_reranker()` routes to `QwenRerankerScorer` when +**either** condition holds: + +* `reranker.model_type == "qwen3"` (the new explicit config value), **or** +* the model name matches `is_qwen3_reranker()`, i.e. contains + `qwen3-reranker` (case-insensitive). + +The name-based leg is not cosmetic โ€” it is load-bearing for the eval harness. +`eval/run_eval.py:191-197` hard-codes `"model_type": "cross-encoder"` in the +config it builds, and `run_eval.py` is out of this task's ownership scope, so +an explicit-config-only route could not have been A/B-tested at all. It also +closes the silent-noise failure above: any Qwen3-Reranker name now gets a +trained scorer instead of a random head, however it is configured. + +The cross-encoder path is untouched and remains the default. `bge-reranker-v2-m3` +still goes `strategy == "rerankers-lib"` โ†’ `Reranker(model_name, +model_type="cross-encoder")`, byte-for-byte the previous behaviour; the bge +re-run below reproduces the baseline first stage exactly, which is the evidence. + +--- + +## 2. Conditions under test + +| | | +|---|---| +| Embedder | `Qwen/Qwen3-Embedding-0.6B` (1024-dim), same as `BASELINE.md`. **Not** the shipped 4B default. | +| Query-side instruction | **Pinned off** via `EMBEDDING_INSTRUCTION=` on every run. A concurrent Phase 1.2 change added a default Qwen3 query prefix (`Given a web search query, retrieve relevant passages that answer the query`) to `RetrievalPipeline._query_instruction()` while this A/B was running; pinning it empty reproduces the `BASELINE.md` first stage and holds the candidate lists identical across all six runs. | +| Index | Rebuilt once at the start, `2026-08-09T06:29:53Z` (docs, 313 chunks) and `06:30:30Z` (mixed, 316 chunks). All six runs reuse that same cached index โ€” verified by the `built_at` timestamps and by md5-hashing the 12 `Documentation/*.md` corpus files before the first run and after the last (**unchanged**). | +| Profile | `PIPELINE_CONFIGS["default"]`, enrichment / overviews / late-chunking / context-expansion / decomposition / verification all OFF, `k = 20`, `chunk_size = 512` โ€” i.e. `run_eval.py` defaults, identical to Phase 0. | +| Hardware | Apple M2 Max, 96 GB, MPS, torch 2.4.1, transformers 4.51.0, rerankers 0.10.0. | +| Load | **Shared.** Other agents were running their own MPS evals against the same GPU throughout. Latency is noisy by construction; every latency-sensitive comparison was run twice and both runs are reported. | + +Model weights on disk: bge-reranker-v2-m3 2.1 GB ยท Qwen3-Reranker-0.6B 1.1 GB ยท +Qwen3-Reranker-4B 7.5 GB (downloaded for this A/B). + +### Deviation from `BASELINE.md`, stated plainly + +The `docs` corpus is live repo content and `Documentation/indexing_pipeline.md` +was edited after the Phase 0 index was built, so the index had to be rebuilt. +Chunk counts came out identical (313 / 316) and the **first stage reproduced +exactly** โ€” mixed nDCG@10 0.805, recall@5 0.917, recall@10 0.972, recall@20 +0.986, matching `BASELINE.md` to three decimals. The bge post-rerank number +moved slightly (mixed 0.903 โ†’ 0.9077, docs 0.731 โ†’ 0.7468) because the reranker +sees slightly different chunk text. **That is why bge was re-run rather than +quoted**; the bge column below, not `BASELINE.md`, is the comparison point. + +--- + +## 3. Quality โ€” `mixed` corpus (n = 72, 316 chunks) โ€” the headline table + +All three rerankers reorder the **same** 20 first-stage candidates per query. + +| Reranker | nDCG@10 first stage | **nDCG@10 post-rerank** | ฮ” vs first stage | recall@5 post-rerank | recall@10 post-rerank | +|---|---|---|---|---|---| +| *(no rerank โ€” first stage)* | 0.805 | โ€” | โ€” | 0.917 | 0.972 | +| `BAAI/bge-reranker-v2-m3` (current default) | 0.805 | **0.9077** | +0.103 | 0.9167 | 0.9861 | +| `Qwen/Qwen3-Reranker-0.6B` | 0.805 | **0.9289** | +0.124 | 0.9583 | 0.9722 | +| `Qwen/Qwen3-Reranker-4B` | 0.805 | **0.9825** | +0.178 | 0.9861 | 0.9861 | + +Deltas against the re-run bge column: **0.6B +0.021 nDCG@10, 4B +0.075 +nDCG@10.** recall@5 post-rerank: 0.6B +0.042, 4B +0.069. recall@10 +post-rerank: 4B unchanged at 0.9861, **0.6B is 0.014 worse** (0.9722 vs +0.9861 โ€” one query out of 72). + +Every quality figure above reproduced **identically** on the second run of each +model (all six runs are deterministic to four decimals). Only latency moved. + +## 4. Quality โ€” all corpora + +`docs` (n = 24, 313 chunks) is the only corpus with real distractors; `atlas7` +(1 chunk) and `hr` (2 chunks) are saturated by construction at k = 20 and test +the plumbing, not the retriever. + +| Corpus | metric | first stage | bge-v2-m3 | Qwen3-0.6B | Qwen3-4B | +|---|---|---|---|---|---| +| `docs` | nDCG@10 | 0.515 | 0.7468 | 0.7868 | **0.9474** | +| `docs` | recall@5 | 0.750 | 0.750 | 0.875 | **0.9583** | +| `docs` | recall@10 | 0.9167 | 0.9583 | 0.9167 | 0.9583 | +| `atlas7` | nDCG@10 | 1.000 | 1.000 | 1.000 | 1.000 | +| `atlas7` | recall@5 / @10 | 1.000 | 1.000 | 1.000 | 1.000 | +| `hr` | nDCG@10 | 0.954 | 1.000 | 1.000 | 1.000 | +| `hr` | recall@5 / @10 | 1.000 | 1.000 | 1.000 | 1.000 | +| `mixed` | nDCG@10 | 0.805 | 0.9077 | 0.9289 | **0.9825** | + +On `docs`, the corpus that actually discriminates: **4B is +0.201 nDCG@10 over +bge** and +0.208 recall@5. The 0.6B is +0.040 nDCG@10 over bge but **loses +recall@10** (0.9167 vs 0.9583) โ€” it demotes one answer-bearing chunk out of the +top 10 that bge keeps in. + +### By dimension (all corpora pooled, n = 144 query evaluations, nDCG@10 post-rerank) + +| Slice | n | bge | Qwen3-0.6B | Qwen3-4B | +|---|---|---|---|---| +| difficulty = easy | 88 | 0.916 | 0.935 | **1.000** | +| difficulty = hard | 56 | 0.904 | 0.920 | **0.955** | +| type = factoid | 72 | 0.936 | 0.941 | **0.983** | +| type = comparative | 18 | 0.955 | 0.957 | **0.983** | +| type = negative | 28 | 0.912 | **0.866** | **0.964** | +| type = procedural | 26 | 0.815 | 0.943 | **1.000** | + +The one slice where a Qwen3 model loses to bge: **0.6B on `negative` questions, +0.866 vs bge's 0.912.** With n = 28 that is roughly one query; treat it as a +diagnostic, not a finding. The 4B wins every slice. + +--- + +## 5. The four known bge regressions โ€” do the Qwen3 models fix them? + +`BASELINE.md` names four queries the bge cross-encoder ranks **worse** than the +first stage did. Per-query nDCG@10 on `mixed`, same candidate lists: + +| Query | first stage | bge | Qwen3-0.6B | Qwen3-4B | verdict | +|---|---|---|---|---|---| +| `atlas7_a16` | 1.000 | 0.431 | **1.000** | **1.000** | **fixed by both** | +| `docs_d15` | 1.000 | 0.316 | 0.631 | **1.000** | fixed by 4B; 0.6B halves the damage but still regresses | +| `docs_d21` | 0.500 | 0.387 | **0.631** | **1.000** | **fixed by both** (both now beat the first stage) | +| `docs_d09` | 0.431 | 0.333 | 0.387 | 0.387 | **shared by all three.** Both Qwen3 models improve on bge but neither reaches the first-stage score | + +**Three of four fixed by the 4B; two of four by the 0.6B. `docs_d09` is a +shared regression โ€” no candidate reranker recovers it.** + +Counting every query on `mixed` whose post-rerank nDCG@10 falls below its +first-stage value: + +| Reranker | queries degraded (of 72) | which | queries improved | +|---|---|---|---| +| bge-v2-m3 | **7** | `atlas7_a16`, `docs_d02`, `d09`, `d14`, `d15`, `d17`, `d21` | 20 | +| Qwen3-0.6B | **5** | `docs_d02`, `d09`, `d15`, `d17`, `d20` | 23 | +| Qwen3-4B | **2** | `docs_d09`, `docs_d17` | 24 | + +The 4B cuts reranker-induced damage from 7 queries to 2. `docs_d09` and +`docs_d17` are the residue every model shares or nearly shares โ€” the honest +statement is that reranker choice does not solve them and they need a different +intervention. + +The `docs` queries `BASELINE.md` lists as retrieved-but-badly-ranked also move, +sharply: + +| Query | first stage | bge | Qwen3-0.6B | Qwen3-4B | +|---|---|---|---|---| +| `docs_d08` (the "clearest case for keeping a cross-encoder") | 0.000 | 0.387 | 0.631 | **1.000** | +| `docs_d07` | 0.387 | 0.387 | 1.000 | **1.000** | +| `docs_d13` | 0.316 | 0.316 | 0.631 | **1.000** | +| `docs_d14` | 0.387 | 0.356 | 1.000 | **1.000** | +| `docs_d19` | 0.333 | 0.387 | 0.631 | **1.000** | +| `docs_d16` | 1.000 | 1.000 | 1.000 | 1.000 | + +`docs_d16` stays at nDCG 1.000 for all three and at recall 0 for all three: it +is a `match: "all"` comparative whose second anchor never enters the candidate +set. **No reranker can fix a first-stage coverage miss** โ€” that one belongs to +Phase 1.2. + +--- + +## 6. Latency โ€” per query, `mixed`, wall clock, MPS, shared GPU + +Every model was run twice on `mixed`. **Both runs are reported; neither is +discarded.** The GPU was shared with other agents' evals throughout, and the +spread between the two runs is the honest measure of how much that matters. + +| Reranker | run | mean | median | **p95** | max | +|---|---|---|---|---|---| +| bge-v2-m3 | 1 | 1883 ms | 1905 ms | 2181 ms | 3675 ms | +| bge-v2-m3 | 2 | 2189 ms | 1625 ms | 3559 ms | 4067 ms | +| Qwen3-0.6B | 1 | 5296 ms | 5578 ms | 6849 ms | 9867 ms | +| Qwen3-0.6B | 2 | 3010 ms | 2836 ms | 4179 ms | 6108 ms | +| Qwen3-4B | 1 | 19520 ms | 14255 ms | 43883 ms | 48064 ms | +| Qwen3-4B | 2 | 11957 ms | 12507 ms | 13238 ms | 14683 ms | + +Reading the two runs together, per 20-candidate query: + +* **bge โ‰ˆ 1.6โ€“2.2 s mean.** Consistent with `BASELINE.md`'s 1.55โ€“1.74 s; the + excess is contention. +* **Qwen3-0.6B โ‰ˆ 3.0โ€“5.3 s mean, ~1.4โ€“2.8ร— bge.** Run 2 is the cleaner + measurement (its p95/median ratio is 1.47 vs run 1's 1.23 on a much higher + base); call it **~1.5โ€“2ร— bge** under light load. +* **Qwen3-4B โ‰ˆ 12โ€“19.5 s mean, ~5.5โ€“10ร— bge.** Run 1's p95 of 43.9 s was taken + while another agent ran a Qwen3-Embedding-4B eval on the same GPU. **Run 2's + 12.0 s mean / 13.2 s p95 is the number to plan against, and it is still + ~5.5ร— bge and ~80ร— the 134โ€“322 ms first stage.** + +Whole-run wall clock, for scale: `--corpus all` (144 query evaluations) took +**210 s** with bge, **575 s** with Qwen3-0.6B, **1916 s** with Qwen3-4B. +The 4B's `mixed`-only run 2 took **875 s** for 72 queries. + +Latency is not a property of the scorer alone: the Qwen3 path pays a full +causal-LM forward pass over `prompt + query + document` per pair, where bge +pays one 512-token encoder pass. Batch size 8 and the 2048-token truncation cap +in `QwenRerankerScorer` are the two knobs; neither was tuned for this A/B, and +tuning them is unmeasured work, not a promise. + +--- + +## 7. Recommendation + +**Adopt Qwen3-Reranker-4B for a quality-first profile; keep +`bge-reranker-v2-m3` as the shipped default until the latency is addressed. +Do not adopt Qwen3-Reranker-0.6B.** + +That is an *adopt-with-size-choice*, and the size choice is 4B-or-nothing. + +**Why 4B is the only Qwen3 worth taking.** It is the largest single measured +quality win in this repo's eval history: **+0.075 nDCG@10 on `mixed` and +0.201 +on `docs`** over a re-run bge baseline, with recall@5 up 0.069 and recall@10 +not regressing. It fixes three of the four known bge regressions, cuts +reranker-induced damage from 7 queries to 2, wins every dimension slice, and +lifts `docs_d08` โ€” the query `BASELINE.md` singles out as the strongest +argument for having a cross-encoder at all โ€” from bge's 0.387 to 1.000. The +roadmap's own bar is "adopt only on a measured win"; this clears it by a wide +margin and is far above the ~2-point threshold below which leaderboard deltas +are known not to transfer. + +**Why it should not become the default today.** ~12 s per query at p95 13.2 s, +against bge's ~2 s, on a single-user, single-threaded `TCPServer`. That is a +user-visible pause on every question and it multiplies under decomposition +(which fans out into multiple rerank calls). It also costs 7.5 GB of weights +resident alongside the embedder and the generation model. The quality is worth +paying for on demand; it is not obviously worth paying for on every message. +Concretely: expose it as an opt-in profile (`reranker.model_name: +"Qwen/Qwen3-Reranker-4B"` in a quality/deep profile), leave `default` and +`fast` on bge, and revisit the default only after the latency work below. + +**Why 0.6B is rejected.** +0.021 nDCG@10 on `mixed` over bge โ€” one to two +queries out of 72, inside the noise band the roadmap itself says not to trust โ€” +bought with a 1.5โ€“2.8ร— latency increase. It also *loses* recall@10 on both +`mixed` (0.9722 vs 0.9861) and `docs` (0.9167 vs 0.9583), and loses to bge on +the `negative` question slice (0.866 vs 0.912). It fixes only two of the four +known regressions. A small ranking gain paid for with a recall loss and 2ร— the +latency is not a win; there is no configuration in which the 0.6B is the right +answer when 4B and bge both exist. + +**What this does not settle.** These numbers are on the 1024-dim +`Qwen3-Embedding-0.6B` index with the query-side instruction pinned off, not on +the shipped 4B embedder and not with the Phase 1.2 instruction prefix on. If +Phase 1.2 changes the embedder default, the first stage changes and this A/B +must be re-run before the reranker choice is final โ€” the 4B's headroom is +largest exactly where the first stage is weakest (`docs`, nDCG 0.515), so a +better first stage will shrink, not grow, its margin. + +### Suggested follow-up before any default flips + +1. **Tune the Qwen3 latency knobs** โ€” batch size (currently 8) and the 2048-token + truncation cap (currently generous: chunks are 512 tokens). Both are + untested; either could move the 12 s materially. +2. **Try `top_k` truncation of the candidate list before reranking.** All the + quality above is on 20 candidates. If reranking the top 10 keeps the nDCG, + it halves the cost. +3. **Re-A/B after Phase 1.2** settles the embedder and the instruction prefix. +4. **`docs_d09` and `docs_d17`** are reranker-proof โ€” every model degrades them. + They are a query-understanding problem, not a reranker problem. + +--- + +## 8. Reproducing every number above + +```bash +cd /path/to/localGPT + +# one-time: fetch the 4B weights (7.5 GB). Disable xet โ€” with it enabled the +# download stalled at 0 bytes twice on this machine (see Caveats). +HF_HUB_DISABLE_XET=1 .venv/bin/python -c \ + "from huggingface_hub import snapshot_download; \ + print(snapshot_download('Qwen/Qwen3-Reranker-4B', max_workers=2))" + +# build/refresh the shared index once, so all three models see identical candidates +EMBEDDING_INSTRUCTION= EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ + .venv/bin/python eval/run_eval.py --corpus all --coverage-only + +# the A/B โ€” run 1 (all corpora) +for M in BAAI/bge-reranker-v2-m3 Qwen/Qwen3-Reranker-0.6B Qwen/Qwen3-Reranker-4B; do + EMBEDDING_INSTRUCTION= EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ + .venv/bin/python eval/run_eval.py --corpus all --reranker "$M" \ + --json-out "eval/results/ab_$(basename $M)_run1.json" +done + +# run 2 โ€” mixed only, for the second latency measurement +for M in BAAI/bge-reranker-v2-m3 Qwen/Qwen3-Reranker-0.6B Qwen/Qwen3-Reranker-4B; do + EMBEDDING_INSTRUCTION= EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ + .venv/bin/python eval/run_eval.py --corpus mixed --reranker "$M" \ + --json-out "eval/results/ab_$(basename $M)_run2.json" +done +``` + +`EMBEDDING_INSTRUCTION=` (empty) is not optional if you want these exact +numbers โ€” without it the Phase 1.2 query prefix changes the first stage. + +`--reranker Qwen/Qwen3-Reranker-*` reaches `QwenRerankerScorer` through the +name-based leg of the loader, because `run_eval.py` hard-codes +`model_type: "cross-encoder"`. In application config, prefer the explicit form: + +```yaml +reranker: + enabled: true + model_type: "qwen3" + model_name: "Qwen/Qwen3-Reranker-4B" + top_k: 10 +``` + +--- + +## 9. Caveats, verbatim + +* **Shared GPU.** Other agents ran MPS evals concurrently for the whole + session. Quality is unaffected (all six runs are bit-identical on every + metric), latency is not. Both runs are published; where they disagree, run 2 + is the lighter-load measurement. +* **The `docs` corpus is live repo content.** It was rebuilt at + `2026-08-09T06:29:53Z` / `06:30:30Z` and md5-verified unchanged across all + six runs, but it is *not* byte-identical to the corpus behind `BASELINE.md` + (`Documentation/indexing_pipeline.md` changed in between). This is why bge was + re-run. +* **`atlas7` and `hr` are saturated** at 1 and 2 chunks. Their 1.000s are + plumbing checks, not retrieval measurements. +* **n = 72 on `mixed`.** A 0.021 nDCG@10 difference is one to two queries. The + 0.6B-vs-bge gap is inside that band; the 4B-vs-bge gap (0.075, and 0.201 on + `docs`) is not. +* **No throughput or memory measurement.** Nothing here says what the 4B does to + concurrent requests or to peak RSS alongside the generation model. +* **The `rerankers` 0.10.0 finding is version-specific.** A later release may add + a Qwen3 backend; the name-based route in the loader would then still win, and + should be revisited at that point. +* **`HF_HUB_DISABLE_XET=1` was required to download the 4B.** With the default + xet transport the two safetensors shards sat at 0 bytes and made no progress + across two attempts; with xet disabled the 7.5 GB fetched in 51 minutes at + ~2.6 MB/s. Recorded because it will cost the next person an hour otherwise. + + +--- + +**Gate correction (2026-08-09, post-adoption):** the header's "`rag_system/main.py` +still ships `BAAI/bge-reranker-v2-m3`" described the tree at measurement time. Shipped +now: default profile reranker disabled; `RERANKER_MODEL` defaults to +`Qwen/Qwen3-Reranker-4B` for the opt-in path. See `eval/DECISIONS.md`. diff --git a/eval/decisions/rfc-shakedown-2026-08-13.md b/eval/decisions/rfc-shakedown-2026-08-13.md new file mode 100644 index 00000000..1b36f2a1 --- /dev/null +++ b/eval/decisions/rfc-shakedown-2026-08-13.md @@ -0,0 +1,860 @@ +# localGPT shakedown on unseen real documents โ€” 23 IETF RFCs (QUIC / HTTP-3) + +Run 2026-08-12 / 2026-08-13 on this Mac. Repo `/Users/prompt/videos/localgpt_08082026/localGPT`, +branch `rearchitect/evidence-gated-aug-2026`, interpreter `.venv/bin/python`, +Ollama at `localhost:11434` (generation `qwen3.5:9b`, enrichment + judge +`qwen3.5:4b`, embedder `microsoft/harrier-oss-v1-0.6b`). Everything local. + +Every number below came from a command executed in this session. Ollama is a +shared local service; the only wall-clock figure offered as evidence is the +indexing time, and even that is "on a machine also running the eval harness". + +> **Read this first.** ยง1-ยง9 are the original run, kept verbatim as the record. +> Finding (1) below was validated and **fixed by the gate on 2026-08-13**, and +> everything downstream of it was re-measured from scratch. The current numbers +> and the current verdict are in **[Post-fix re-run](#post-fix-re-run-2026-08-13)**, +> after ยง8. Finding (2) is unchanged by the fix. + +**Short answer to "does the setup work on unseen real documents?" โ€” no, not yet. +Two things break before retrieval quality is even the question:** + +1. **The chunker silently discards ~48% of the corpus.** Any document over + ~10,000 markdown tokens loses roughly half its text before it is ever + embedded. 10 of 24 mechanically verified gold rows have their answer in *no* + indexed chunk as a result. This is a bug in shipped code, it affects + `Documentation/design_rationale.md` in the existing `docs` corpus too, and it + is the single largest finding here. +2. **The cross-reference extractor resolves 0 of 731 references** on real + documents. Not "few" โ€” zero. The extraction half of roadmap 4.2 is inert on + any corpus whose filenames are not literal substrings of its own prose. + +Retrieval, measured only over the rows whose content actually reached the index, +is respectable (recall@10 0.929, nDCG@10 0.601). End-to-end answer quality is +poor (4/24 judged pass), and the dominant failure mode is the system saying +"there is no mention of that in the provided text" โ€” which is what you would +expect from a half-indexed corpus, and is at least an honest failure rather than +a hallucinated one. + +--- + +## 1. Corpus + +`eval/corpora/rfc/` โ€” **23 documents, 1,511,267 bytes (1.44 MiB)**, downloaded +verbatim from `https://www.rfc-editor.org/rfc/rfcNNNN.txt`. Reproducible with +`.venv/bin/python eval/corpora/rfc/download.py`; `--check` re-verifies sizes and +the link graph. Full per-file table, rationale and exclusions: +`/Users/prompt/videos/localgpt_08082026/localGPT/eval/corpora/rfc/MANIFEST.md`. + +The cluster is the QUIC / HTTP-3 family plus what it is defined against: + +* QUIC core โ€” 8999, 9000, 9001, 9002, 9221, 9369, 9308, 9312 +* HTTP over QUIC โ€” 9114, 9204, 9218, 9220, 9297, 9298, 9412, and the HTTP/2 + counterparts 8336, 8441 that those documents defer their semantics to +* shared normative dependencies โ€” 2119, 8174, 8126, 6066, 7301 + +Selection rule, enforced mechanically by `download.py --check`: **every document +references or is referenced by at least two others in the set.** Measured: +**110 directed intra-corpus `RFC NNNN` references**; the lowest-degree document +(RFC 6066) touches 4 others; RFC 9000 is cited by 14 of the other 22. + +RFC 5234 (ABNF) was in the first draft and was removed when `--check` reported +it at **degree 0** โ€” the family's ABNF references all route through RFC 9110 / +9112, which are excluded for budget. RFC 8126 replaced it. + +Excluded for the ~1.5 MB budget, and this is a real limitation rather than a +neutral trim: RFC 9110 (502,941 B โ€” the most-cited document in the HTTP/3 +sub-cluster), RFC 8446 (337,736 B โ€” RFC 9001's principal TLS dependency), +9113, 6455, 7541, 9111, 9112. The 9001โ†”8446 cross-reference pair the task +suggested is therefore *not* in the gold set; the TLS-side crossref rows anchor +on 7301 and 6066 instead. + +--- + +## 2. Gold set + +`eval/goldset/rfc.jsonl` โ€” **24 rows, hand-authored**, same schema as +`eval/goldset/acquisition.jsonl` plus the same `expected_sources` / +`anchor_doc` / `multi_document` / `requires_crossref` fields. Answer-bearing +anchors are catalogued in `eval/corpora/rfc/rfc.facts.json` (26 facts). + +Gate: `eval/verify_rfc_goldset.py`, modelled on `eval/verify_crossref_goldset.py`. +Output, verbatim: + +``` +/Users/prompt/videos/localgpt_08082026/localGPT/eval/goldset/rfc.jsonl: 24 rows, 26 expected strings, 23 source documents + expected_in_source 26/26 + no_verbatim_leak 26/26 + unique_to_source 26/26 + fact_id_resolves 26/26 + crossref_flag_consistent 24/24 + multi_document_consistent 24/24 + requires_crossref=true 10 + multi_document=true 2 + question_type {'factoid': 16, 'negative': 3, 'procedural': 3, 'comparative': 2} + difficulty {'easy': 9, 'hard': 15} + documents referenced 16/23 + +all row-level checks passed. +``` + +Notes on the gates: + +* Comparison is whitespace-normalised and case-insensitive, because RFCs are + hard-wrapped at 72 columns and a sentence-length anchor necessarily spans a + line break. `run_eval.py` scores chunk relevance with exactly the same + normalisation (`run_eval.norm`), so a string that passes here is a string the + metric can match. +* **`unique_to_source` passed 26/26, so nothing had to be exempted.** The + mechanism for RFC boilerplate exists in the script (`UNIQUENESS_EXEMPT`, empty + and documented as such) because the obvious anchors โ€” the BCP 14 sentence, the + bare phrase "MUST NOT" โ€” occur in all 23 documents and were deliberately + avoided during authoring. Every anchor that shipped carries a hex constant, a + codepoint, a named parameter or a distinctively worded restriction. +* The 10 `requires_crossref=true` rows each state document A's premise and ask + for a fact whose text exists only in document B โ€” e.g. RFC 9002's deferral of + the anti-amplification limit to RFC 9000 ยง8.1 (`rfc_q15`), RFC 9412's deferral + of ORIGIN payload semantics to RFC 8336 (`rfc_q22`), RFC 9220's reuse of RFC + 8441's `:protocol` pseudo-header (`rfc_q21`). +* 2 rows are `multi_document` with `match: "all"` (`rfc_q18` needs both QUIC v1 + and v2 initial salts; `rfc_q24` needs both HTTP/3's requirement and RFC 9000's + parameter definition). +* 16 of the 23 documents carry at least one anchor. The other 7 (2119, 8174, + 9308, 9204's siblings etc.) are present as distractors and reference targets. + +**Gate 2 (reachability after chunking) FAILS for 10 of 24 rows.** That is not a +gold-set defect โ€” see ยง4 โ€” and the rows are deliberately left in place so the +gate keeps reporting the loss. + +--- + +## 3. Indexing with product defaults + +`build_product_index.py` โ€” `PIPELINE_CONFIGS["default"]` unchanged except for +storage location (scratch; `eval/.eval_indexes/` and `index_store/` untouched) +and `chunk_size = 512`. Confirmed active at run time: contextual enrichment ON +(window 1, `qwen3.5:4b`), document overviews ON + embedded sidecar, late +chunking ON, `extract_crossrefs` ON, docling chunker. + +| Stage | Wall clock | +|---|---:| +| Document processing + chunking + 23 document overviews | 201.8 s | +| Cross-reference extraction | 0.26 s | +| Contextual enrichment, 387 chunks | 1763.8 s | +| Embedding generation | 52.4 s | +| Late-chunk embedding & indexing | 65.6 s | +| Overview embedding (23 docs โ†’ `.vectors.npz`) | 1.5 s | +| **Total** | **2089.2 s (34.8 min)** | + +**387 chunks** from 23 files. Enrichment is 84% of the build and runs at 0.22 +chunks/s on the shared Ollama. + +**No `Ollama likely FRONT-TRUNCATED` warning occurred** โ€” checked in the index +build log, the E2E run log and the judge log. Count is 0 in all three. + +### Cross-reference extraction on real documents โ€” the headline + +Printed by the pipeline itself, identical in the harness-config build and the +product-defaults build: + +``` +๐Ÿ”— Cross-references: 731 reference(s) in 260 chunk(s); 0 resolved to 0 document(s). +``` + +**0 of 731 resolved.** `crossref_diagnostic.py` re-runs `annotate_chunks` +unmodified over the same 387 chunk texts under three document-naming schemes +(the schemes are passed as different `known_documents` id sets โ€” a public +argument; nothing in `rag_system/` was patched): + +| Naming scheme | refs | of which `document` mentions | resolved | documents linked | +|---|---:|---:|---:|---:| +| shipped `RFC 9000 - QUIC A UDP-Based ....txt` | 731 | 0 | **0** | 0 | +| `RFC 9000.txt` | 822 | 91 | **91** | 18 | +| `QUIC A UDP-Based ... - RFC 9000.txt` (citation word order) | 766 | 35 | **35** | 11 | + +Three distinct reasons, all measured: + +1. **The document-mention pass requires the whole normalised filename to appear + in the text.** Under the shipped naming, **0 of 23** document names occur + anywhere in the corpus (0 occurrences). Under `RFC 9000.txt`, 20 of 23 names + occur, 178 times. RFCs cite each other as `RFC 9000` or `[RFC9000]`, never as + `RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport`. The + numeric-prefix-stripping alias in `CrossRefExtractor.__init__` does not help, + because these filenames start with `RFC`, not with digits. +2. **96% of what the extractor does find is unusable.** The 731 references break + down as **703 `section` + 28 `exhibit` + 0 `document`**. A `section` ref is + recorded as a bare `"section 8.1"` with `target_doc: None`, because no + document is named `section 8 1`. In an RFC corpus that is the wrong shape + entirely: of **1,713** `Section N` occurrences, **355 are explicitly + qualified** โ€” `Section 8.1 of [QUIC-TRANSPORT]` โ€” and the extractor's + `_SECTION_RE` captures the section number and **throws away the `[โ€ฆ]` + qualifier, which is precisely the hop target.** Those 355 are the real + cross-reference graph of this corpus and the extractor cannot see any of it. +3. **The qualifiers are symbolic, not numeric.** 52 distinct tags appear in + those qualified references; the most common are `[QUIC-TRANSPORT]` (74), + `[HTTP]` (44), `[QUIC-TLS]` (39), `[QUIC]` (20), `[RFC9000]` (17). Resolving + them means reading each document's own References section to map + `[QUIC-TRANSPORT] โ†’ RFC 9000`. Filename matching cannot do it in principle, + only the 17 literal `[RFC9000]`-style tags are reachable that way. + +The `MAX_CROSSREFS_PER_CHUNK = 8` cap is a minor contributor, not the cause: 8 +chunks hit the cap, and 745 label+section references were found before it +against 731 kept. It would matter more under a naming scheme that produced +document mentions, since the document pass runs last and gets the leftovers. + +**Reading:** this is not "the extractor scored low", it is "the extractor is +inert on any corpus we did not name for it". The index-time half of roadmap 4.2 +has, until now, only ever been run on `eval/corpora/acquisition/`, whose +filenames (`05_escrow_agreement.pdf`) were written to match the prose ("the +Escrow Agreement"). The corpus and the extractor were co-designed. On documents +that were not, it produces nothing to hop on. + +--- + +## 4. The chunker discards half the corpus + +*(Fixed by the gate on 2026-08-13 โ€” see [P1](#p1-the-chunker-fix-verified-independently). +This section records the pre-fix state and the diagnosis.)* + +This was found by gate 2 and is the reason most other numbers on this page are +depressed. + +`run_eval.py --coverage-only` on the 387-chunk index: + +``` +โš ๏ธ 10 gold row(s) whose expected text is in NO indexed chunk: + rfc_q02, rfc_q09, rfc_q10, rfc_q11, rfc_q13, + rfc_q14, rfc_q15, rfc_q17, rfc_q18, rfc_q24 +``` + +Cause, isolated in `chunker_loss_repro.py`, measured three ways: + +* **The LanceDB table holds 52% of the corpus's whitespace-normalised + characters** (790,786 of 1,511,267 raw bytes' worth). Per document the + retention splits cleanly in two: documents under ~10,000 markdown tokens + retain ~100%, documents over it retain 45-57%. RFC 9114 retains 0.45, RFC 9000 + 0.50, RFC 9297 1.02 (the >1 is the one-sentence overlap duplicating text). +* **The conversion step is innocent.** `DocumentConverter.convert_to_markdown` + on RFC 9114 returns 139,294 normalised characters against 139,286 in the + source โ€” ratio 1.00, and the missing gold string is present in its output. +* **The loss is in `MarkdownRecursiveChunker._split_text`** + (`rag_system/ingestion/chunking.py`). `re.split(f'({sep})', chunk)` yields + `[t0, sep, t1, sep, t2, sep, t3]`; the re-combining loop advances `i += 3` + after a match, which lands on a separator rather than the next body segment. + It emits `[sep+t1, sep, sep+t3]`: `t0` is never emitted and `t2` is replaced + by a bare separator. Synthetic repro โ€” 8 paragraphs in, `max_chunk_size=200` + tokens: **segments 1,3,5,7 kept; segments 0,2,4,6 lost; 0.50 of characters + survive.** + +It only fires when a chunk exceeds `max_chunk_size`, and `DoclingChunker` +constructs its internal `MarkdownRecursiveChunker` with `max_chunk_size=10_000`. +That is why no previous corpus exposed it: **12 of 23 RFCs cross the threshold, +against 1 of 13 files in `Documentation/`.** + +**The pre-existing `docs` corpus is affected**, in exactly one file: +`Documentation/design_rationale.md`, 13,285 tokens, **0.49 retained**. Every +other Documentation file is under the threshold at ~1.00. So the `docs` and +`mixed` numbers in `eval/BASELINE.md` were measured against a corpus that is +missing half of one of its thirteen files. Small, but it means the baseline will +move when this is fixed, and it should not be treated as a regression. + +Nothing was edited to establish any of this โ€” the repro imports the shipped +classes and calls them. + +--- + +## 5. Retrieval + +Harness configuration (`run_eval.py` machinery, driven from +`run_rfc_eval.py`, which injects the `rfc` corpus and redirects the index root +into scratch without editing anything under `eval/`): enrichment / overviews / +late chunking / context expansion OFF, reranker OFF, `k = 20`, +`chunk_size = 512`, embedder `microsoft/harrier-oss-v1-0.6b`, hybrid search. +`--retry off` for the comparison cells, per the determinism protocol. + +The `final == first_stage` invariant held on all 24 queries (rerank off, hop +off), so the first-stage numbers are the whole story. + +| Slice | n | recall@5 | recall@10 | recall@20 | nDCG@10 | 1st ms mean | +|---|---:|---:|---:|---:|---:|---:| +| **all rows** | 24 | 0.417 | 0.542 | 0.583 | **0.392** | 282 | +| `requires_crossref=true` | 10 | 0.500 | 0.600 | 0.600 | 0.518 | 341 | +| control (`=false`) | 14 | 0.357 | 0.500 | 0.571 | 0.303 | 239 | +| **reachable rows only** (gate 2 pass) | 14 | 0.714 | **0.929** | **1.000** | **0.601** | 276 | +| โ”œ `requires_crossref=true` | 6 | 0.833 | 1.000 | 1.000 | 0.696 | 325 | +| โ”” control | 8 | 0.625 | 0.875 | 1.000 | 0.529 | 240 | +| unreachable rows (chunker loss) | 10 | 0.000 | 0.000 | 0.000 | 0.100 | 289 | + +The 0.100 nDCG on the unreachable slice is entirely `rfc_q18`, a `match: "all"` +row that scores nDCG 1.000 on the one of its two anchors that survived chunking +while scoring recall 0 โ€” the coverage-vs-ranking split `eval/README.md` warns +about, showing up for real. + +By dimension (all 24 rows, pooled): easy `recall@10` 0.556 / nDCG 0.380, hard +0.533 / 0.400; factoid 0.625 / 0.361, procedural 0.667 / 0.667, negative 0.333 / +0.210, comparative 0.000 / 0.500. + +With `--retry on` (the shipped profile) the numbers are slightly *worse* โ€” +all-rows recall@10 0.500 vs 0.542, nDCG@10 0.355 vs 0.392 โ€” and the retry fires +on 6/24 queries (`q01, q03, q04, q05, q17, q19`) with 5 rewrites kept. Per-query +latency rises from 282 ms to 1480 ms. On 24 rows this is a diagnostic, not a +verdict on the retry; but it is the opposite sign to what the retry is for, and +the rewrites are nondeterministic, which is why the comparison cells above use +`--retry off`. + +### Unseen-real vs authored-synthetic + +Against `eval/BASELINE.md` ยง *Phase 4 baseline* (same embedder, same `k`, same +chunk size, reranker off): + +| Corpus | n | chunks | recall@5 | recall@10 | recall@20 | nDCG@10 | +|---|---:|---:|---:|---:|---:|---:| +| `acq` (10 synthetic M&A PDFs) | 24 | 13 | 0.958 | 1.000 | 1.000 | 0.810 | +| `acq+docs` | 48 | 373 | 0.917 | 0.958 | 1.000 | 0.738 | +| `acq+docs`, crossref slice | 11 | 373 | 1.000 | 1.000 | 1.000 | 0.748 | +| **`rfc`, all rows** | 24 | 387 | 0.417 | 0.542 | 0.583 | 0.392 | +| **`rfc`, reachable rows** | 14 | 387 | 0.714 | 0.929 | 1.000 | 0.601 | +| **`rfc`, reachable crossref** | 6 | 387 | 0.833 | 1.000 | 1.000 | 0.696 | + +The honest comparison is the reachable-rows line, since the all-rows line is +mostly measuring the chunker bug. On that line: + +* **Recall is comparable.** 0.929 @10 and 1.000 @20 against `acq+docs`'s 0.958 + and 1.000, on a corpus with a similar chunk count (387 vs 373) but 5.7x the + text and genuine same-domain distractors rather than topically disjoint ones. +* **Ranking is meaningfully worse.** nDCG@10 0.601 vs 0.738. The answer-bearing + chunk is usually retrieved but sits at rank 3-5 instead of 1-2. That is the + expected effect of same-domain distractors: 23 documents that all discuss + connection IDs, frame types and settings compete far harder than localGPT + documentation competes with an M&A deal room. +* **The crossref slice is again not the weak one** โ€” 1.000 recall@10 and 0.696 + nDCG against the control's 0.875 / 0.529. This reproduces the `acq` finding + and for the same reason: a crossref query that names document A *and* asks + about B's subject matter hands the hybrid FTS leg lexical signal from both + ends. It is a property of how honest crossref questions are worded, not + evidence that pointer-following is solved. Note the control slice here is + genuinely harder than on `acq` (0.529 vs 0.796), which is the same-domain + distractor effect. + +--- + +## 6. End-to-end judged answers + +All 24 gold questions through the in-process `Agent` on the product-defaults +index, `force_rag=True`, `agent._query_cache.clear()` before every query. +Answers recorded verbatim in `results/rfc_e2e_answers.jsonl`. + +Per-query wall clock: mean 88.7 s, median 59.4 s, min 41.0 s, max 465.0 s +(`rfc_q21`), total 2130 s. Zero exceptions; zero truncation warnings. + +Judged with `eval/judge.py`'s `GroundednessJudge` on `qwen3.5:4b`, prompt v1, +**k = 5 votes per row**, with the mapping the task specifies: +`EVIDENCE = the system answer`, `ANSWER = the gold answer`, +`QUESTION = the gold query`. + +| Slice | n | majority pass (โ‰ฅ3/5) | unanimous pass | unanimous fail | mean k/5 | cited an expected source | +|---|---:|---:|---:|---:|---:|---:| +| all | 24 | **4** | 2 | 17 | 0.92 | 24 / 24 | +| single-doc | 14 | **3** | 1 | 11 | 0.86 | 14 / 14 | +| `requires_crossref` | 10 | **1** | 1 | 6 | 1.00 | 10 / 10 | +| โ€” reachable rows | 14 | 3 | โ€” | โ€” | 1.21 | 14 / 14 | +| โ€” unreachable rows | 10 | 1 | โ€” | โ€” | 0.50 | 10 / 10 | + +`k_true` histogram: `{0: 17, 1: 1, 2: 2, 3: 1, 4: 1, 5: 2}` โ€” the judge is +mostly unanimous, in the negative. + +Per row: + +| id | k/5 | reachable | crossref | first 100 chars of the system answer | +|---|---:|---|---|---| +| rfc_q01 | 0/5 | yes | | "there is **no mention of an `ack_delay_exponent` parameter**โ€ฆ" | +| rfc_q02 | 0/5 | no | | "a QUIC endpoint is generally free to advertise any valueโ€ฆthere is **no expliโ€ฆ**" | +| rfc_q03 | 0/5 | yes | | "there is **no single specific error code** defined solely forโ€ฆ" | +| rfc_q04 | **3/5** | yes | | "if a QUIC sender has not yet measured any RTT samplesโ€ฆ" | +| rfc_q05 | **4/5** | yes | | "The sender is permitted to send up to two probe datagramsโ€ฆ" | +| rfc_q06 | 0/5 | yes | | "Based on RFC 9297 โ€ฆ and RFC 9003 (the specification forโ€ฆ" | +| rfc_q07 | **5/5** | yes | | "the version field value for long headers isโ€ฆ" | +| rfc_q08 | 0/5 | yes | | "there is **no specific numerical range**โ€ฆ" | +| rfc_q09 | 0/5 | no | | "**no**, a โ€ฆ DoQ client should not open its initial connection using UDP port 53โ€ฆ" | +| rfc_q10 | 0/5 | no | | "there is **no specific numeric value or fixed amount** of flow-control creditโ€ฆ" | +| rfc_q11 | 0/5 | no | | "the receiving endpoint treats the missing parameter as having a value of **0**" | +| rfc_q12 | 0/5 | yes | | "there are **no specific header field names**โ€ฆ" | +| rfc_q13 | 0/5 | no | | "the default applied is considered low โ€ฆ between **0 and 6**" | +| rfc_q14 | 0/5 | no | | "there is **no mention** of a \"latency spin bit\"โ€ฆ" | +| rfc_q15 | **5/5** | no | Y | "if no additional data โ€ฆ (three times) โ€ฆ" | +| rfc_q16 | 0/5 | yes | Y | "Based on **RFC 9000 (QUIC)** and its associated transport parametersโ€ฆ" | +| rfc_q17 | 0/5 | no | Y | "there is no specific mention of the **width** (in bits)โ€ฆ" | +| rfc_q18 | 0/5 | no | Y | "The provided context does not specify two distinct saltsโ€ฆ" | +| rfc_q19 | 0/5 | yes | Y | "there is **no mention** of a specific error code or numeric valueโ€ฆ" | +| rfc_q20 | 2/5 | yes | Y | "mandates two additional demands beyond expert reviewโ€ฆ" | +| rfc_q21 | 2/5 | yes | Y | "first defined in **RFC 8441**, published in September 2018โ€ฆ" | +| rfc_q22 | 0/5 | yes | Y | "there is no \"HTTP/3 ORIGIN extension\" that utilizes a corresponding HTTP/2 frameโ€ฆ" | +| rfc_q23 | 1/5 | yes | Y | "there are no direct facts linking Connect-UDP tunnels to a specific capsule typeโ€ฆ" | +| rfc_q24 | 0/5 | no | Y | "QUIC does not have a single fixed transport parameterโ€ฆ" | + +Two observations that are not judge artefacts: + +* **Citations are perfect and answers are not.** All 24 answers cited at least + one chunk from a document that genuinely holds the answer. The retrieval stage + got the endpoint to the right document 24/24 times, and the synthesis stage + still concluded "not stated here" 15 times. That gap is the chunker: the right + document was retrieved with the wrong half of its text in it. +* **The system fails safe, not loudly.** The dominant failure is an explicit + "there is no mention of X in the provided text", not a fabricated value. One + clear hallucination: `rfc_q06` cites "RFC 9003", which does not exist. + `rfc_q13` states a wrong range (0-6, default "low" instead of 0-7, default 3). + +### Judge-suspect rows โ€” for a Sonnet-voter rerun, not adjudicated here + +Flagged mechanically by `judge_e2e.py` (verdict/reason inconsistency, or a split +vote), the known `qwen3.5:4b` failure mode from +`eval/decisions/phase4-escalation-rerun.md` ยง6: + +| row | k/5 | why it is suspect | +|---|---:|---| +| `rfc_q04` | 3/5 | split vote, no majority worth trusting | +| `rfc_q15` | 5/5 | vote 3's reason is phrased as a rejection while the verdict is `true` | +| `rfc_q20` | 2/5 | split vote **and** vote 5 votes `true` with a reason my screen reads as negative | +| `rfc_q21` | 2/5 | split vote **and** vote 1 votes `true` with a reason my screen reads as negative | + +A second, separate mechanical screen (`results/orientation_flags.txt`) โ€” rows +the judge failed even though the system answer contains โ‰ฅ60% of the gold +answer's distinctive tokens (hex codes, 3+-digit numbers, snake_case +identifiers, ALLCAPS names): **`rfc_q02`, `rfc_q09`, `rfc_q21`, `rfc_q22`, +`rfc_q23`.** These are candidates for the same rerun. I have not decided any of +them. Two are worth the gate's attention specifically because the *orientation* +may be doing the work rather than the judge: + +* `rfc_q09` โ€” the system answer states "clients MUST establish a QUIC connection + to UDP port 853" and that port 53 should not be used, hedged with "unless + there is a specific mutual agreement". Every one of the 5 votes rejected the + gold answer's phrase "explicitly forbidden" as unsupported by that hedge. +* `rfc_q02` โ€” the system answer asserts the *opposite* of the gold ("no explicit + floor"), so the 0/5 looks correct; it is on the screen only because it repeats + the parameter name. The screen over-flags by construction and is offered as a + filter, not a finding. + +**Caveat on the judging orientation, stated because it changes how the 4/24 +should be read.** With `EVIDENCE = system answer` and `ANSWER = gold answer`, +the question being asked is "does the system's answer contain everything the +gold answer asserts?" โ€” a recall check on the system answer, not a groundedness +check. A system answer that is correct but states less than the gold answer +fails. This is the mapping the task specified and I ran it as specified, but the +4/24 is a lower bound on correctness, not an estimate of it. + +--- + +## 7. Anomalies and things that did not go as expected + +1. **387 chunks for 1.44 MiB.** Noticed as "suspiciously few" before gate 2 ran; + it was the first symptom of ยง4. +2. **`--retry on` measured worse than `--retry off`** on this corpus (nDCG@10 + 0.355 vs 0.392, recall@10 0.500 vs 0.542) while costing 5x the latency. 24 + rows; a diagnostic only. +3. **`rfc_q21` took 465 s** โ€” 5x the median. It is the row whose answer needed + RFC 8441; the agent decomposed it and ran several sub-queries. +4. **No `FRONT-TRUNCATED` warnings at all**, across the 34.8-minute enrichment + build (387 enrichment calls), the 24-query E2E run and 120 judge calls. +5. **`rfc_q06`'s answer cites a non-existent "RFC 9003".** The only outright + fabrication in 24 answers. +6. The gold set passed all six row-level gates on the first run, with no string + needing a uniqueness exemption โ€” which is worth recording because RFC prose + is far more repetitive than the synthetic corpora, and I expected to need + exemptions. + +## 8. Verdict, and the three weakest spots + +*(Pre-fix verdict, superseded by [P6](#p6-revised-verdict).)* + +**Does the setup work on unseen real documents? Not today.** The retrieval stage +is fine โ€” on the content that reaches the index it finds the right document +every time and the right chunk 93% of the time at k=10. Everything downstream is +undermined by content that never got indexed, and the one Phase-4 mechanism this +corpus was built to exercise does not fire at all. + +The three weakest spots, named: + +1. **`MarkdownRecursiveChunker._split_text` drops ~half of any document over + 10,000 tokens.** Highest severity, cheapest fix, and it invalidates the + `docs`/`mixed` baseline slightly (`design_rationale.md` at 0.49 retention). + A background task has been filed with the repro and the measurement. Nothing + else on this page should be re-measured until it is fixed. +2. **Cross-reference resolution is filename-shaped and real corpora are not.** + 0/731 resolved. The fix that would actually pay here is not a better filename + heuristic โ€” it is keeping the `[โ€ฆ]` qualifier that `_SECTION_RE` currently + discards (355 explicit cross-document section references in this corpus) and + resolving symbolic reference tags through each document's own References + section. Until then, treat the index-time half of roadmap 4.2 as + demonstrated only on the corpus it was co-designed with. +3. **Synthesis gives up rather than degrading.** 15 of 24 answers are "the + provided text does not mention this", including on rows where retrieval + ranked the answer-bearing document first. Some of that is (1); some of it is + a 9b model being asked to answer from chunks that carry an + enrichment-generated preamble plus dense RFC prose. Worth re-running once (1) + is fixed, because right now the two causes cannot be separated. + +Honourable mention, not in the top three because it is a measurement property +rather than a defect: **the `requires_crossref` slice is once again the *easier* +slice** (reachable rows: recall@10 1.000 vs 0.875, nDCG 0.696 vs 0.529). Two +corpora now agree. A crossref gold row that names document A and asks about B's +subject matter is not a hard retrieval problem, because it leaks lexical signal +from both ends. If Phase 4.2 needs a discriminative first-stage metric, the gold +set that provides it has to ask about the *pointer* and nothing about the +target's content โ€” and that gold set still does not exist. + +--- + +# Post-fix re-run (2026-08-13) + +The gate validated the ยง4 finding and fixed +`MarkdownRecursiveChunker._split_text` in `rag_system/ingestion/chunking.py` +(separator now reattached to the segment that follows it; `seg0` emitted). +Everything in ยง3-ยง6 was re-measured from scratch against the fixed chunker: +both scratch indexes deleted and rebuilt, gold set and corpus untouched. The +pre-fix outputs are preserved under `prefix_results/` and as +`results/*.prefix.jsonl`. + +I did not edit `rag_system/` โ€” the diff is the gate's. My only change to my own +tooling was adding resume support to `judge_e2e.py` after the judging process +was killed at row 17 by a harness timeout; the 17 completed rows were kept and +the remaining 7 judged on restart, same model, same prompt, same k. + +## P1. The chunker fix, verified independently + +`chunker_loss_repro.py` re-run unchanged: + +| Measurement | pre-fix | post-fix | +|---|---:|---:| +| Synthetic repro: segments kept (8 paragraphs, `max_chunk_size=200`) | `[1,3,5,7]` | **`[0,1,2,3,4,5,6,7]`** | +| Synthetic repro: character ratio | 0.50 | **1.00** | +| `rfc` corpus, characters retained through convert โ†’ chunk | **0.52** | **1.02** | +| RFC 9000 (93,395 tokens) | 0.50 | 1.02 | +| RFC 9114 (38,505 tokens) | 0.45 | 1.02 | +| `Documentation/design_rationale.md` (13,285 tokens) | 0.49 | 1.09 | + +The >1.00 ratios are the chunker's one-sentence overlap duplicating text at +chunk boundaries, which was always there; the 12 RFCs over the 10,000-token +threshold now behave exactly like the 11 under it. + +## P2. Index rebuild and gate 2 + +| | pre-fix | post-fix | +|---|---:|---:| +| Harness-config index (no LLM) | 387 chunks, 106.2 s | **683 chunks, 142.4 s** | +| Product-defaults index | 387 chunks, 2089.2 s (34.8 min) | **683 chunks, 3096.7 s (51.6 min)** | +| โ€” document processing + 23 overviews | 201.8 s | 163.5 s | +| โ€” contextual enrichment | 1763.8 s | 2764.2 s | +| โ€” embedding / late-chunk / overview-embed | 52.4 / 65.6 / 1.5 s | 91.4 / 71.9 / 1.3 s | +| **Gate 2 โ€” gold rows reachable in an indexed chunk** | **14 / 24** | **24 / 24** | + +`gold coverage 24/24 rows reachable in the index`, `coverage_failures` empty. +The 76% index-time increase is 76% more chunks to enrich, at the same +0.25 chunks/s. Ollama is shared with nothing else during the build, but it is a +local service and this is one measurement on one machine. + +**No `Ollama likely FRONT-TRUNCATED` warning in any post-fix log** โ€” index +build, E2E run, judge run. Count 0 in all three, as before. + +## P3. Cross-reference extraction โ€” unchanged, and now measured on the whole corpus + +``` +๐Ÿ”— Cross-references: 1403 reference(s) in 485 chunk(s); 0 resolved to 0 document(s). +``` + +| | pre-fix (387 chunks) | post-fix (683 chunks) | +|---|---:|---:| +| references extracted | 731 | **1403** | +| of which `section` / `exhibit` / `document` | 703 / 28 / 0 | **1334 / 69 / 0** | +| **resolved** | **0** | **0** | +| documents linked | 0 | 0 | +| resolved under `RFC 9000.txt` naming | 91 (18 docs) | 124 (20 docs) | +| resolved under `<Title> - RFC 9000.txt` naming | 35 (11 docs) | 53 (12 docs) | +| document names occurring anywhere in corpus text (shipped naming) | 0 / 23 | **0 / 23** | +| chunks hitting `MAX_CROSSREFS_PER_CHUNK` | 8 | 23 | + +The finding is unchanged and is now measured against the complete corpus rather +than half of it: **0 of 1403.** Every word of ยง3's diagnosis stands, including +the 355 `Section N of [TAG]` qualified cross-document references whose `[TAG]` +`_SECTION_RE` discards, and the 52 symbolic reference tags +(`[QUIC-TRANSPORT]`, `[QUIC-TLS]`, โ€ฆ) that no filename heuristic can resolve. + +## P4. Retrieval, before โ†’ after + +Identical configuration both times: `run_eval.py` machinery, enrichment / +overviews / late chunking / context expansion OFF, reranker OFF, `k = 20`, +`chunk_size = 512`, `--retry off`, `microsoft/harrier-oss-v1-0.6b`, hybrid +search. `final == first_stage` invariant held on all 24 queries in both runs. + +| Slice | n | recall@5 | recall@10 | recall@20 | nDCG@10 | +|---|---:|---|---|---|---| +| **all rows** | 24 | 0.417 โ†’ **0.750** | 0.542 โ†’ **0.833** | 0.583 โ†’ **0.958** | 0.392 โ†’ **0.659** | +| `requires_crossref=true` | 10 | 0.500 โ†’ **0.600** | 0.600 โ†’ **0.800** | 0.600 โ†’ **1.000** | 0.518 โ†’ **0.719** | +| control (`=false`) | 14 | 0.357 โ†’ **0.857** | 0.500 โ†’ **0.857** | 0.571 โ†’ **0.929** | 0.303 โ†’ **0.616** | +| rows unreachable pre-fix | 10 | 0.000 โ†’ **0.700** | 0.000 โ†’ **0.800** | 0.000 โ†’ **1.000** | 0.100 โ†’ **0.723** | + +"Reachable-only" is no longer a separate cut โ€” all 24 rows pass gate 2, so that +slice *is* the corpus. The right pre-fix comparison for the honest reader is +therefore the pre-fix reachable-only line, and even against that the fix is a +clear win on ranking: + +| | n | recall@5 | recall@10 | recall@20 | nDCG@10 | +|---|---:|---:|---:|---:|---:| +| pre-fix, reachable rows only | 14 | 0.714 | 0.929 | 1.000 | 0.601 | +| **post-fix, all rows** | 24 | 0.750 | 0.833 | 0.958 | **0.659** | + +recall@10 dips (0.929 โ†’ 0.833) because the post-fix figure is over 10 harder +rows that previously could not be scored at all, not because anything regressed; +recall@20 is 0.958 with the two misses being `rfc_q19` (recall 0 at @5/@10, 1 at +@20) and `rfc_q18` (a `match: "all"` row that needs both salts). + +By dimension, post-fix: procedural 1.000 recall@10 / 1.000 nDCG, factoid 0.875 / +0.634, negative 0.667 / 0.421, comparative 0.500 / 0.702; easy 0.889 / 0.623, +hard 0.800 / 0.680. + +Against the authored-synthetic baseline (`eval/BASELINE.md`, same embedder, same +`k`, reranker off): + +| Corpus | n | chunks | recall@5 | recall@10 | recall@20 | nDCG@10 | +|---|---:|---:|---:|---:|---:|---:| +| `acq+docs` | 48 | 373 | 0.917 | 0.958 | 1.000 | 0.738 | +| `acq+docs`, crossref slice | 11 | 373 | 1.000 | 1.000 | 1.000 | 0.748 | +| **`rfc`, post-fix, all rows** | 24 | 683 | 0.750 | 0.833 | 0.958 | **0.659** | +| **`rfc`, post-fix, crossref slice** | 10 | 683 | 0.600 | 0.800 | 1.000 | **0.719** | + +Unseen-real is still measurably harder than authored-synthetic โ€” recall@10 +0.833 vs 0.958, nDCG@10 0.659 vs 0.738 โ€” but the gap is now a retrieval gap of +the size you would expect from same-domain distractors, not the collapse the +pre-fix numbers showed. The crossref slice is within 0.03 nDCG of the `acq` +crossref slice. + +With `--retry on` the retry now fires on **1/24** queries (was 6/24) โ€” better +first-pass evidence means less to retry โ€” and scores slightly below `--retry +off` (nDCG@10 0.617 vs 0.659, recall@10 0.792 vs 0.833), the same sign as +pre-fix. 24 rows; a diagnostic, not a verdict on the retry. + +## P5. End-to-end judged answers, before โ†’ after + +Same protocol both times: in-process `Agent`, product defaults, `force_rag=True`, +`_query_cache.clear()` per query, `qwen3.5:4b` judge, prompt v1, **k = 5**, +`EVIDENCE = system answer`, `ANSWER = gold answer`, verifier suffix stripped by +`judge.py` as before. Zero exceptions in either run. + +| Slice | n | majority pass (โ‰ฅ3/5) | unanimous pass | unanimous fail | mean k/5 | cited an expected source | +|---|---:|---|---|---|---|---| +| all | 24 | 4 โ†’ **3** | 2 โ†’ 2 | 17 โ†’ 19 | 0.92 โ†’ 0.67 | 24/24 โ†’ **24/24** | +| single-doc | 14 | 3 โ†’ **0** | 1 โ†’ 0 | 11 โ†’ 13 | 0.86 โ†’ 0.14 | 14/14 โ†’ 14/14 | +| `requires_crossref` | 10 | 1 โ†’ **3** | 1 โ†’ 2 | 6 โ†’ 6 | 1.00 โ†’ 1.40 | 10/10 โ†’ 10/10 | + +`k_true` histogram: `{0:17, 1:1, 2:2, 3:1, 4:1, 5:2}` โ†’ `{0:19, 1:1, 2:1, 3:1, 5:2}`. +Per-query wall clock: mean 88.7 โ†’ 83.5 s, median 59.4 โ†’ 58.6 s, max 465.0 โ†’ +222.2 s, total 2130 โ†’ 2003 s. + +Per row: + +| id | pre | post | ฮ” | crossref | was unreachable pre-fix | +|---|---:|---:|---:|---|---| +| rfc_q01 | 0/5 | 0/5 | 0 | | | +| rfc_q02 | 0/5 | 0/5 | 0 | | yes | +| rfc_q03 | 0/5 | 0/5 | 0 | | | +| rfc_q04 | 3/5 | 0/5 | **โˆ’3** | | | +| rfc_q05 | 4/5 | 0/5 | **โˆ’4** | | | +| rfc_q06 | 0/5 | 0/5 | 0 | | | +| rfc_q07 | 5/5 | 2/5 | **โˆ’3** | | | +| rfc_q08 | 0/5 | 0/5 | 0 | | | +| rfc_q09 | 0/5 | 0/5 | 0 | | yes | +| rfc_q10 | 0/5 | 0/5 | 0 | | yes | +| rfc_q11 | 0/5 | 0/5 | 0 | | yes | +| rfc_q12 | 0/5 | 0/5 | 0 | | | +| rfc_q13 | 0/5 | 0/5 | 0 | | yes | +| rfc_q14 | 0/5 | 0/5 | 0 | | yes | +| rfc_q15 | 5/5 | 5/5 | 0 | Y | yes | +| rfc_q16 | 0/5 | 0/5 | 0 | Y | | +| rfc_q17 | 0/5 | 0/5 | 0 | Y | yes | +| rfc_q18 | 0/5 | 0/5 | 0 | Y | yes | +| rfc_q19 | 0/5 | 0/5 | 0 | Y | | +| rfc_q20 | 2/5 | 1/5 | โˆ’1 | Y | | +| rfc_q21 | 2/5 | 3/5 | **+1** | Y | | +| rfc_q22 | 0/5 | 0/5 | 0 | Y | | +| rfc_q23 | 1/5 | 5/5 | **+4** | Y | | +| rfc_q24 | 0/5 | 0/5 | 0 | Y | yes | + +**4 โ†’ 3 is not a real difference at n = 24 with a 4b judge, and I am not +claiming one.** Three things in this table *are* worth reading, and all three are +checkable against the recorded answers: + +1. **Refusals halved.** Answers that open by saying the corpus does not contain + the fact: **11/24 โ†’ 6/24** (`q01, q03, q08, q17, q23` all stopped refusing; + `q10, q12, q14, q18, q19, q22` still refuse). That is the chunker fix showing + up in synthesis, and it is the effect the fix was supposed to have. +2. **The crossref slice improved and the single-doc slice did not.** crossref + 1/10 โ†’ 3/10 (mean k 1.00 โ†’ 1.40); single-doc 3/14 โ†’ 0/14 (mean k 0.86 โ†’ + 0.14). `rfc_q23` is the cleanest single win in the whole exercise: pre-fix it + answered "no direct facts linking Connect-UDP tunnels to a specific capsule + type" and attributed the capsule to a non-existent "RFC 9076"; post-fix it + answers "Capsule Type 0x00 (the DATAGRAM Capsule) โ€ฆ Section 3.5", 5/5. +3. **Three single-doc rows genuinely got worse, and it is not a judge flip.** + Reading the recorded answers: + * `rfc_q04` (3/5 โ†’ 0/5): pre-fix the answer stated "**SHOULD be set to 333 + milliseconds**". Post-fix it cites RFC 9002 ยง5.3 (`smoothed_rtt`/`rttvar`) + and says RTT is treated as infinite until an acknowledgement arrives โ€” + never giving 333 ms. More indexed text pulled synthesis to a different, + also-relevant part of the same document. All 5 judge reasons say exactly + that. + * `rfc_q05` (4/5 โ†’ 0/5): post-fix the answer says "one or two probe + datagrams", quoting RFC 9002's ยง6.2 overview sentence, and drops the "MUST + send at least one ack-eliciting packet" clause the pre-fix answer quoted. + The judge rejects the gold answer's "ack-eliciting" and "full-sized" on + that basis. + * `rfc_q07` (5/5 โ†’ 2/5): the answer still gives the right value, + **0x6b3343cf**, but adds an unsupported explanation of how the constant was + derived; 3 of 5 votes reject on the explanation, not the value. The + product's own verifier flagged it too โ€” the answer carries + `[Confidence: 90%] [Warning: Low confidence. Groundedness: False]`, which + `judge.py` strips before judging. + +**Hallucinated citations persist and did not improve.** Answers citing RFC +numbers absent from the corpus: 8/24 pre-fix, 9/24 post-fix. Some are real +documents the corpus text itself references (6298 for TCP's RTO, 6455, 7540) and +are fair; others are plain misattributions โ€” pre-fix `rfc_q23` credited the +DATAGRAM capsule to "RFC 9076", post-fix `rfc_q06` credits the QUIC datagram +extension to "RFC 9287" instead of RFC 9221. Every answer in both runs carries +the verifier suffix, so the verifier is running; it is not preventing these. + +### Judge-suspect rows โ€” for a Sonnet-voter rerun, not adjudicated here + +| run | verdict/reason inconsistency or split vote | orientation screen (judged fail, โ‰ฅ60% of gold's distinctive tokens present) | +|---|---|---| +| pre-fix | `rfc_q04`, `rfc_q15`, `rfc_q20`, `rfc_q21` | `rfc_q02`, `rfc_q09`, `rfc_q21`, `rfc_q22`, `rfc_q23` | +| **post-fix** | **`rfc_q07`, `rfc_q13`, `rfc_q21`** | **`rfc_q02`, `rfc_q07`, `rfc_q16`, `rfc_q22`** | + +Post-fix detail: `rfc_q07` 2/5 and `rfc_q21` 3/5 are split votes with no +majority worth trusting; `rfc_q13` has one vote whose reason my screen reads as +positive against a `false` verdict. `rfc_q07` appears on both lists โ€” it is the +single strongest candidate for a stronger voter, since the disputed content is +an added explanation rather than the answer itself. I have adjudicated none of +them. + +The orientation caveat from ยง6 still applies unchanged and matters more now that +refusals have halved: with `EVIDENCE = system answer` and `ANSWER = gold`, an +answer that is correct but says *less* than the gold answer, or *more*, fails. +`rfc_q05` and `rfc_q07` are both of that shape. **3/24 is a lower bound on +correctness, not an estimate of it.** + +## P6. Revised verdict + +**Retrieval on unseen real documents now works; answer synthesis and +cross-reference extraction do not.** + +The chunker fix moved the corpus from half-indexed to fully indexed +(gate 2 14/24 โ†’ 24/24, character retention 0.52 โ†’ 1.02) and retrieval moved with +it: recall@20 0.583 โ†’ 0.958, nDCG@10 0.392 โ†’ 0.659, and the crossref slice to +1.000 recall@20 / 0.719 nDCG โ€” within 0.03 nDCG of the synthetic `acq` corpus it +was designed against. Unseen-real remains harder than authored-synthetic +(nDCG@10 0.659 vs 0.738), which is the honest size of the same-domain-distractor +penalty and is a normal number, not a failure. + +The two weakest spots, revised: + +1. **Cross-reference extraction is still inert: 0 of 1403 references resolved.** + Unchanged by the fix and now measured on the complete corpus. The index-time + half of roadmap 4.2 produces nothing to hop on for any corpus whose filenames + are not literal substrings of its own prose. The payoff is not a better + filename heuristic but keeping the `[TAG]` qualifier `_SECTION_RE` currently + discards (355 qualified cross-document section references here) and resolving + symbolic tags through each document's References section. +2. **Synthesis is now the bottleneck, and it is unstable.** 3/24 judged pass + under a strict orientation, refusals down but not gone (6/24), hallucinated + RFC attributions flat (9/24 answers), and three single-doc rows that + regressed because *more* correct retrieved text moved the answer to a + different, also-relevant part of the right document (`rfc_q04`, `rfc_q05`, + `rfc_q07`). That last pattern is the one worth chasing: it says the failure + is context selection and answer composition, not retrieval. + +Third, unchanged from ยง8 and still true: **the `requires_crossref` slice is not +the hard slice.** Post-fix it beats its own control on nDCG (0.719 vs 0.616) and +matches it on recall, and it is the only slice whose E2E pass count improved +(1/10 โ†’ 3/10). Two corpora now agree. A first-stage metric that discriminates +for Phase 4.2 needs a gold set that asks about the *pointer* and nothing about +the target's content, and that gold set still does not exist. + +--- + +## 9. Files produced + +New files under the repo (for the gate to review and commit): + +| Path | What | +|---|---| +| `eval/corpora/rfc/*.txt` | the 23 downloaded RFCs, unmodified | +| `eval/corpora/rfc/download.py` | reproducible download + link-graph check | +| `eval/corpora/rfc/MANIFEST.md` | per-file manifest, selection rationale, exclusions, naming note | +| `eval/corpora/rfc/rfc.facts.json` | 26 answer-bearing anchors (kept inside `rfc/` so `verify_facts.py`'s `corpora/*.facts.json` glob does not pick it up and no existing gate output changes) | +| `eval/goldset/rfc.jsonl` | 24 verified gold rows | +| `eval/verify_rfc_goldset.py` | the row-level gate | + +Nothing under `rag_system/`, `backend/`, `Documentation/` or any pre-existing +`eval/` file was modified. + +Scratch (`.../scratchpad/rfc_shakedown/`): + +| Path | What | +|---|---| +| `build_rfc_goldset.py` | authors the gold set + facts sidecar | +| `run_rfc_eval.py` | injects the `rfc` corpus into `run_eval.py` without editing it | +| `build_product_index.py` | product-defaults index build | +| `run_e2e.py`, `judge_e2e.py` | E2E answers and k=5 judging | +| `crossref_diagnostic.py`, `chunker_loss_repro.py`, `slice_results.py` | the three diagnostics | +| `compare_e2e.py` | before/after table for the judged E2E pass | +| `results/rfc_retrieval_retryoff_postfix.json`, `..._retryon_postfix.json` | **post-fix** retrieval runs | +| `results/rfc_e2e_answers.jsonl`, `results/rfc_e2e_judged.jsonl`, `results/rfc_e2e_judge_summary.json` | **post-fix** answers, all 120 votes, summary | +| `results/*.prefix.jsonl`, `prefix_results/` | the complete pre-fix outputs, preserved unmodified | +| `results/crossref_diagnostic.json`, `results/chunker_loss.json` | diagnostics, re-run post-fix | +| `results/orientation_flags.txt`, `results/orientation_flags_postfix.txt` | the orientation screen, both runs | +| `product_index/` | the throwaway product-defaults index (683 chunks post-fix, overviews, latechunk) | + +### Reproducing + +```bash +cd /Users/prompt/videos/localgpt_08082026/localGPT +SCR=/private/tmp/claude-501/-Users-prompt-videos-localgpt-08082026/4d62420b-7ab2-4be1-90f2-708d7bae9146/scratchpad/rfc_shakedown + +.venv/bin/python eval/corpora/rfc/download.py --check # corpus + link graph +.venv/bin/python eval/verify_rfc_goldset.py # gold-set gate +.venv/bin/python $SCR/run_rfc_eval.py --corpus rfc --coverage-only # gate 2 +.venv/bin/python $SCR/run_rfc_eval.py --corpus rfc --retry off \ + --json-out $SCR/results/rfc_retrieval_retryoff.json # retrieval +.venv/bin/python $SCR/slice_results.py $SCR/results/rfc_retrieval_retryoff.json +.venv/bin/python $SCR/crossref_diagnostic.py +.venv/bin/python $SCR/chunker_loss_repro.py +.venv/bin/python $SCR/build_product_index.py # ~35 min +.venv/bin/python $SCR/run_e2e.py # ~36 min +.venv/bin/python $SCR/judge_e2e.py # ~3 min +``` + +--- + +## Gate validation and final judged numbers (2026-08-13) + +The chunker data-loss claim was independently reproduced at the gate (RFC 9000 +retention 49.12% pre-fix; alternating-section synthetic repro) before the fix +(commit 7d71051) was written; the fix is property-tested lossless. The gold-set +gates were re-run at the gate (all pass). Crossref inertness was independently +reproduced (0 resolved on real RFC text). Retrieval was re-run live at the +gate: recall@20 0.958 exactly matches; recall@5/@10 and nDCG@10 reproduced +within one query of the agent's numbers (0.708/0.792/0.617 vs +0.750/0.833/0.659) โ€” small tie-break variance, direction unchanged. + +**Final E2E answer quality (Sonnet subagent panel, 3 voters, the validated +judge โ€” supersedes the 4b numbers above):** **5/24 pass** โ€” single-doc 1/14, +requires_crossref 4/10. All 24 rows unanimous across voters; agreement with +the 4b k=5 majorities on 22/24, and both disagreements are rows the 4b flagged +judge-suspect (`rfc_q07`, `rfc_q20` โ€” Sonnet rules both grounded). Per-row +votes in `rfc-shakedown-sonnet-panel.json`. + +The bottleneck is therefore answer synthesis on dense unseen technical text, +not retrieval: the right documents are retrieved (recall@20 0.958) and cited +(24/24), but the 9b generation model frequently answers from its own prior +instead of the supplied snippets โ€” e.g. `rfc_q01` asserts a default +`ack_delay_exponent` of -1 and fabricates a supporting quote attributed to +"RFC 9002 Section 13.4" while the correct value (3) sat in the retrieved text. +The verifier correctly marked most of these low-confidence, but the answer +text still leads with the wrong claim. diff --git a/eval/decisions/rfc-shakedown-sonnet-panel.json b/eval/decisions/rfc-shakedown-sonnet-panel.json new file mode 100644 index 00000000..86027c1d --- /dev/null +++ b/eval/decisions/rfc-shakedown-sonnet-panel.json @@ -0,0 +1,266 @@ +{ + "rfc_q01": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q02": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q03": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q04": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q05": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q06": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q07": { + "sonnet": [ + true, + true, + true + ], + "sonnet_maj": true, + "fourb_k5": 2, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q08": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q09": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q10": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q11": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q12": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q13": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q14": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": false + }, + "rfc_q15": { + "sonnet": [ + true, + true, + true + ], + "sonnet_maj": true, + "fourb_k5": 5, + "fourb_maj": true, + "requires_crossref": true + }, + "rfc_q16": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": true + }, + "rfc_q17": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": true + }, + "rfc_q18": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": true + }, + "rfc_q19": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": true + }, + "rfc_q20": { + "sonnet": [ + true, + true, + true + ], + "sonnet_maj": true, + "fourb_k5": 1, + "fourb_maj": false, + "requires_crossref": true + }, + "rfc_q21": { + "sonnet": [ + true, + true, + true + ], + "sonnet_maj": true, + "fourb_k5": 3, + "fourb_maj": true, + "requires_crossref": true + }, + "rfc_q22": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": true + }, + "rfc_q23": { + "sonnet": [ + true, + true, + true + ], + "sonnet_maj": true, + "fourb_k5": 5, + "fourb_maj": true, + "requires_crossref": true + }, + "rfc_q24": { + "sonnet": [ + false, + false, + false + ], + "sonnet_maj": false, + "fourb_k5": 0, + "fourb_maj": false, + "requires_crossref": true + } +} \ No newline at end of file diff --git a/eval/decisions/synthesis-ab-arm-e-panel.json b/eval/decisions/synthesis-ab-arm-e-panel.json new file mode 100644 index 00000000..121cd276 --- /dev/null +++ b/eval/decisions/synthesis-ab-arm-e-panel.json @@ -0,0 +1,218 @@ +{ + "rfc_q01": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q02": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q03": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q04": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q05": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q06": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q07": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": false + }, + "rfc_q08": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q09": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q10": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q11": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q12": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q13": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q14": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q15": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q16": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q17": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q18": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q19": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q20": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q21": { + "votes": [ + true, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q22": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q23": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q24": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + } +} \ No newline at end of file diff --git a/eval/decisions/synthesis-ab-arm-f-panel.json b/eval/decisions/synthesis-ab-arm-f-panel.json new file mode 100644 index 00000000..61754fe9 --- /dev/null +++ b/eval/decisions/synthesis-ab-arm-f-panel.json @@ -0,0 +1,375 @@ +{ + "arm": "F", + "description": "arm C strict prompt + cross-leg dedupe + 12k-token synthesis context budget", + "date": "2026-08-14", + "protocol": "v1 prompts, verifier suffix stripped, 3 blind Sonnet subagent voters, 2-of-3 majority", + "majority_pass": 16, + "n": 24, + "single_doc_pass": 10, + "single_doc_n": 14, + "crossref_pass": 6, + "crossref_n": 10, + "splits": 1, + "rows": [ + { + "id": "rfc_q01", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The answer restates the evidence's default value of 3 and multiplier of 8 exactly.", + "The answer's default value of 3 and multiplier of 8 both appear verbatim in the evidence.", + "Evidence states the same default of 3 and multiplier of 8 that the answer reports." + ] + }, + { + "id": "rfc_q02", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence explicitly states no floor value was found, so the claimed floor of 2 is not supported.", + "Evidence explicitly states no minimum/floor was found, but the answer asserts a floor of 2 that is not in the evidence.", + "Evidence explicitly states no minimum/floor value was found, contradicting the answer's claim of a floor of 2." + ] + }, + { + "id": "rfc_q03", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The answer's error code name and value 0x10 match the evidence exactly.", + "The error code name, hex value, and description all match the evidence exactly.", + "Evidence states NO_VIABLE_PATH is error code 0x10, matching the answer exactly." + ] + }, + { + "id": "rfc_q04", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence states the information could not be found, so the 333ms/kInitialRtt claim is unsupported.", + "Evidence states the information could not be found, but the answer supplies a specific 333ms value not present in evidence.", + "Evidence states the information could not be found, so the 333ms kInitialRtt figure is unsupported." + ] + }, + { + "id": "rfc_q05", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The answer matches the evidence's statement of at least one ack-eliciting packet and up to two full-sized datagrams.", + "Both the 'at least one ack-eliciting packet' and 'up to two full-sized datagrams' claims are stated verbatim in the evidence.", + "Evidence states at least one ack-eliciting packet and up to two full-sized datagrams, matching the answer." + ] + }, + { + "id": "rfc_q06", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives the type values and bit pattern but never identifies the low bit as the LEN bit, an added claim.", + "The evidence gives the type values 0x30/0x31 and bit pattern but never states the low bit is the LEN bit, an added unsupported claim.", + "Evidence gives the 0x30/0x31 type values and bit pattern but never states the low bit is the LEN bit, an unsupported addition." + ] + }, + { + "id": "rfc_q07", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The version number 0x6b3343cf matches the evidence exactly with no added claims.", + "The evidence directly states the QUIC v2 long header version field value is 0x6b3343cf, matching the answer.", + "Evidence directly states the version 2 header value is 0x6b3343cf, matching the answer." + ] + }, + { + "id": "rfc_q08", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The 0-255 byte range and comparison to version 1's 20-byte cap both match the evidence.", + "The evidence's 0-255 byte range and version 1's 20-byte cap support the answer's version-independent characterization.", + "Evidence states the field is 0-255 bytes in general and 20 bytes max for version 1, matching the answer's comparison." + ] + }, + { + "id": "rfc_q09", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence confirms port 853 by default and explicitly forbids port 53, matching the answer.", + "The port 853 default and prohibition of port 53 are both explicitly stated in the evidence.", + "Evidence confirms port 853 as default and explicitly forbids port 53, matching the answer." + ] + }, + { + "id": "rfc_q10", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The 1,024-byte flow-control credit figure matches the evidence exactly.", + "The 1,024-byte flow-control credit figure is stated verbatim in the evidence.", + "Evidence states at least 1,024 bytes of flow-control credit per unidirectional stream, matching the answer." + ] + }, + { + "id": "rfc_q11", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence explicitly disclaims a general default and never gives the identifier 0x01, both added by the answer.", + "Evidence explicitly states this information was not found, while the answer asserts a specific zero default and identifier 0x01 not present in evidence.", + "Evidence explicitly states it lacks information on the default capacity and never mentions identifier 0x01, so the answer's claims are unsupported." + ] + }, + { + "id": "rfc_q12", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The three header fields listed match the evidence exactly.", + "All three header fields (Content-Length, Content-Type, Transfer-Encoding) are listed exactly in the evidence.", + "Evidence lists exactly Content-Length, Content-Type, and Transfer-Encoding as disqualifying header fields." + ] + }, + { + "id": "rfc_q13", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The range 0-7 and default of 3 with smaller values as higher precedence match the evidence.", + "The 0-7 range and default of 3, with smaller values meaning higher precedence, match the evidence.", + "Evidence states urgency ranges 0-7 with smaller values as higher precedence and a default of 3, matching the answer." + ] + }, + { + "id": "rfc_q14", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The bit position matches the evidence's description of the third-most-significant bit of byte 0.", + "The third-most-significant-bit location of the first octet is stated directly in the evidence.", + "Evidence states the spin bit is the third-most-significant bit of the first octet, matching the answer." + ] + }, + { + "id": "rfc_q15", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The three-times-received-data limit matches the evidence's anti-amplification description.", + "The three-times anti-amplification limit is stated verbatim in the evidence.", + "Evidence states the anti-amplification limit is three times the data received, matching the answer." + ] + }, + { + "id": "rfc_q16", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The PTO formula in the answer is stated verbatim in the evidence.", + "The PTO formula given in the answer is quoted directly from the evidence.", + "The formula quoted in the answer appears verbatim in the evidence text describing PTO computation." + ] + }, + { + "id": "rfc_q17", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence does not mention the tag being computed 'over the Retry pseudo-packet,' an added claim.", + "The evidence supports the 128-bit/AEAD_AES_128_GCM claims but never mentions 'the Retry pseudo-packet', an unsupported addition.", + "Evidence states the tag is AEAD_AES_128_GCM output but never specifies it is computed over the Retry pseudo-packet, an unsupported addition." + ] + }, + { + "id": "rfc_q18", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "Both salt values match but the evidence never attributes the original salt to RFC 9001, an added claim.", + "The salt values match, but the answer's attribution of the original salt to 'RFC 9001' is not stated anywhere in the evidence.", + "Evidence never names the original salt's source document as RFC 9001, only as 'the QUIC-TLS document,' so that identifier is an unsupported addition." + ] + }, + { + "id": "rfc_q19", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The alert number 120 matches the evidence exactly.", + "The numeric alert value 120 is stated directly in the evidence.", + "Evidence states the TLS alert number is 120, matching the answer." + ] + }, + { + "id": "rfc_q20", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The requirement for a formal, permanent, readily available specification on top of expert review matches the evidence.", + "The requirement for a formal, permanent, readily available specification alongside expert review is stated in the evidence.", + "Evidence describes Specification Required as demanding a formal, permanent, readily available public specification beyond expert review, matching the answer." + ] + }, + { + "id": "rfc_q21", + "requires_crossref": true, + "votes": [ + false, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence ties the pseudo-header to an Extended CONNECT request specifically, while the answer broadens this to the CONNECT method generally.", + "RFC 8441 as the origin of :protocol and its use on Extended CONNECT/HEADERS requests are both supported by the evidence.", + "Evidence confirms :protocol was first defined in RFC 8441 and appears on request HEADERS within an Extended CONNECT request, matching the answer." + ] + }, + { + "id": "rfc_q22", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence explicitly states the payload content is not specified, so the Origin-Entry blocks claim is unsupported.", + "Evidence explicitly states the payload content is not specified, but the answer claims it holds 'Origin-Entry blocks', an unsupported addition.", + "Evidence explicitly states the payload content is not specified, so the answer's claim of Origin-Entry blocks is unsupported." + ] + }, + { + "id": "rfc_q23", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The DATAGRAM Capsule Type value 0x00 matches the evidence exactly.", + "The DATAGRAM Capsule Type value 0x00 is stated directly in the evidence.", + "Evidence defines the DATAGRAM Capsule Type as value 0x00, matching the answer." + ] + }, + { + "id": "rfc_q24", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The identifier and stream requirement match, but the evidence never attributes the parameter to RFC 9000, an added claim.", + "The parameter name and 0x09 identifier match, but the evidence never attributes the parameter definition to 'RFC 9000', an unsupported addition.", + "Evidence never states the transport parameter is defined in RFC 9000, so that identifier is an unsupported addition." + ] + } + ] +} \ No newline at end of file diff --git a/eval/decisions/synthesis-ab-arm-g-panel.json b/eval/decisions/synthesis-ab-arm-g-panel.json new file mode 100644 index 00000000..265ee039 --- /dev/null +++ b/eval/decisions/synthesis-ab-arm-g-panel.json @@ -0,0 +1,385 @@ +{ + "arm": "G", + "description": "arm F + Qwen3-Reranker-4B final-stage selection: union-of-max min_score 0.5 over original+sub-queries, min_keep 3, top_k 10, scored on core chunk text", + "date": "2026-08-15", + "protocol": "v1 prompts, verifier suffix stripped, 3 blind Sonnet subagent voters, 2-of-3 majority", + "majority_pass": 18, + "n": 24, + "single_doc_pass": 11, + "single_doc_n": 14, + "crossref_pass": 7, + "crossref_n": 10, + "splits": 1, + "vs_arm_f": { + "gains": [ + "rfc_q02", + "rfc_q17", + "rfc_q22" + ], + "losses": [ + "rfc_q20" + ] + }, + "rows": [ + { + "id": "rfc_q01", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The answer's default value of 3 and multiplier of 8 both appear verbatim in the evidence.", + "The answer's default value of 3 and multiplier of 8 are both stated verbatim in the evidence.", + "Evidence states the default is 3 with a multiplier of 8, exactly matching the answer." + ] + }, + { + "id": "rfc_q02", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly states the parameter MUST be at least 2, matching the answer's floor claim.", + "The evidence explicitly states the value MUST be at least 2, matching the answer's floor claim.", + "Evidence explicitly states active_connection_id_limit MUST be at least 2, matching the answer." + ] + }, + { + "id": "rfc_q03", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence names NO_VIABLE_PATH with code 0x10 exactly as the answer states.", + "The evidence names NO_VIABLE_PATH with code 0x10 exactly as the answer states.", + "Evidence gives NO_VIABLE_PATH with code 0x10, matching the answer exactly." + ] + }, + { + "id": "rfc_q04", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives the 333ms value but never uses the identifier \"kInitialRtt\", which the answer adds without support.", + "The evidence gives the 333ms value but never names it kInitialRtt, so the answer adds an identifier not present in the evidence.", + "The evidence states 333ms but never names the value kInitialRtt, so the answer adds an unstated identifier." + ] + }, + { + "id": "rfc_q05", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states both the MUST-send-one and MAY-send-up-to-two-datagrams claims verbatim.", + "The evidence states at least one ack-eliciting packet and up to two full-sized datagrams, matching the answer exactly.", + "Evidence states at least one ack-eliciting packet and up to two full-sized datagrams, matching the answer." + ] + }, + { + "id": "rfc_q06", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives the type values and bit pattern but never states that the low bit is the LEN bit, an unsupported addition.", + "The evidence gives the type values 0x30/0x31 but never states that the low bit is called the LEN bit, an unsupported added claim.", + "Evidence gives the type values and bit pattern but never states the low bit is the LEN bit, an added unstated claim." + ] + }, + { + "id": "rfc_q07", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the QUIC v2 version field value 0x6b3343cf exactly as the answer does.", + "The evidence states the version 2 value is 0x6b3343cf, matching the answer exactly.", + "Evidence states the version 2 field is 0x6b3343cf, matching the answer exactly." + ] + }, + { + "id": "rfc_q08", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The 0-255 byte range and the comparison to version 1's stated 20-byte cap are both directly derivable from the evidence.", + "The evidence states the field is between 0 and 255 bytes generally and MUST NOT exceed 20 bytes in version 1, matching the answer's contrast.", + "Evidence gives the 0-255 byte general range and the 20-byte version 1 cap, matching the answer's comparison." + ] + }, + { + "id": "rfc_q09", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence confirms port 853 as default and explicitly forbids port 53 for DoQ, matching the answer.", + "The evidence states DoQ uses port 853 by default and MUST NOT use port 53, matching the answer.", + "Evidence states DoQ uses port 853 by default and MUST NOT use port 53, matching the answer." + ] + }, + { + "id": "rfc_q10", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the 1,024-byte SHOULD credit per unidirectional stream exactly as the answer claims.", + "The evidence states at least 1,024 bytes of flow-control credit per unidirectional stream, matching the answer.", + "Evidence states at least 1,024 bytes of flow-control credit per unidirectional stream, matching the answer." + ] + }, + { + "id": "rfc_q11", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence supports the zero-default claim but never mentions identifier \"0x01\" or that no dynamic table entries may be inserted, which the answer adds.", + "The evidence never gives the setting identifier 0x01 nor states that no dynamic table entries may be inserted, both unsupported additions.", + "Evidence never gives the identifier 0x01 and describes a 0-RTT exception the answer's blanket 'zero' claim ignores." + ] + }, + { + "id": "rfc_q12", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence lists exactly these three header fields as making a message ineligible for the Capsule Protocol.", + "The evidence lists exactly Content-Length, Content-Type, and Transfer-Encoding as disqualifying headers, matching the answer.", + "Evidence lists Content-Length, Content-Type, and Transfer-Encoding as disqualifying headers, matching the answer." + ] + }, + { + "id": "rfc_q13", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the 0-7 range, default of 3, and that 0 is highest/7 is lowest urgency, all matching the answer.", + "The evidence states the default urgency is 3 and the range is 0 to 7 with 0 as highest, matching the answer.", + "Evidence states urgency ranges 0-7 with 0 highest and a default of 3, matching the answer." + ] + }, + { + "id": "rfc_q14", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the spin bit is the third-most-significant bit of the first octet of the short header, matching the answer exactly.", + "The evidence states the spin bit is the third-most-significant bit of the first octet, matching the answer exactly.", + "Evidence states the spin bit is the third-most-significant bit of the first octet, matching the answer." + ] + }, + { + "id": "rfc_q15", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the three-times anti-amplification limit exactly as the answer claims.", + "The evidence states no more than three times the data received from the unvalidated address, matching the answer.", + "Evidence states the anti-amplification limit is three times the data received, matching the answer." + ] + }, + { + "id": "rfc_q16", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The answer quotes the PTO formula exactly as given in the evidence without adding unsupported claims.", + "The evidence gives the exact formula PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay, matching the answer.", + "Evidence's headline formula matches the answer's PTO calculation exactly." + ] + }, + { + "id": "rfc_q17", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the 128-bit tag computed via AEAD_AES_128_GCM over the Retry pseudo-packet, matching the answer.", + "The evidence states the tag is 128 bits computed with AEAD_AES_128_GCM over the Retry pseudo-packet contents, matching the answer.", + "Evidence states the Integrity Tag is 128 bits computed via AEAD_AES_128_GCM over the Retry pseudo-packet, matching the answer." + ] + }, + { + "id": "rfc_q18", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives only the version 2 salt value; the version 1 salt value stated in the answer does not appear anywhere in the evidence.", + "The evidence gives only the version 2 salt value and never states the original version 1 salt value the answer claims.", + "Evidence gives only the version 2 salt value and never states the version 1 salt's hex value, which the answer fabricates." + ] + }, + { + "id": "rfc_q19", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly assigns alert description 120 to no_application_protocol, matching the answer.", + "The evidence states the numeric alert description is 120, matching the answer.", + "Evidence states the no_application_protocol alert number is 120, matching the answer." + ] + }, + { + "id": "rfc_q20", + "requires_crossref": true, + "votes": [ + false, + false, + true + ], + "majority_pass": false, + "reasons": [ + "The evidence says informal documentation outside the RFC path is also accepted, contradicting the answer's claim that the specification must be \"formal\".", + "The evidence explicitly allows informal documentation outside the RFC path, contradicting the answer's claim that a formal specification is required.", + "Evidence describes the extra requirement of a permanent, readily available public specification beyond expert review, matching the answer." + ] + }, + { + "id": "rfc_q21", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence confirms :protocol was first defined in RFC 8441 and may appear on CONNECT request HEADERS, matching the answer.", + "The evidence states the :protocol field originates in RFC 8441 and may appear on request HEADERS for CONNECT, matching the answer.", + "Evidence states :protocol was defined in RFC 8441 and appears on CONNECT request HEADERS, matching the answer." + ] + }, + { + "id": "rfc_q22", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states frame type 0x0c (12) with a payload of zero or more Origin-Entry fields, matching the answer.", + "The evidence states the frame type is 0x0c (12) and carries zero or more Origin-Entry fields, matching the answer.", + "Evidence states the HTTP/2 ORIGIN frame is type 0x0c carrying Origin-Entry fields, matching the answer." + ] + }, + { + "id": "rfc_q23", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence defines the DATAGRAM (0x00) Capsule Type exactly as the answer states.", + "The evidence states the DATAGRAM Capsule Type has value 0x00, matching the answer exactly.", + "Evidence states the DATAGRAM Capsule Type has value 0x00, matching the answer." + ] + }, + { + "id": "rfc_q24", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence never states that initial_max_streams_uni is defined in RFC 9000, an unsupported addition in the answer.", + "The evidence never attributes initial_max_streams_uni's definition to RFC 9000, an unsupported added claim.", + "Evidence names initial_max_streams_uni and identifier 0x09 but never attributes it to RFC 9000, an added unstated claim." + ] + } + ] +} \ No newline at end of file diff --git a/eval/decisions/synthesis-ab-arm-g2-panel.json b/eval/decisions/synthesis-ab-arm-g2-panel.json new file mode 100644 index 00000000..72febf37 --- /dev/null +++ b/eval/decisions/synthesis-ab-arm-g2-panel.json @@ -0,0 +1,399 @@ +{ + "arm": "G2", + "description": "composer decomposition (per-SQ full pipeline + compose), forced via compose_sub_answers=True", + "shared_conditions": "arm G reranker selection + temperature-0 decomposition; identical deterministic row sets (6 decomposed queries, same sub-queries verbatim)", + "date": "2026-08-15", + "protocol": "v1 prompts, verifier suffix stripped, 3 blind Sonnet subagent voters, 2-of-3 majority", + "majority_pass": 17, + "n": 24, + "single_doc_pass": 11, + "crossref_pass": 6, + "decomposed_pass": 3, + "splits": 0, + "rows": [ + { + "id": "rfc_q01", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The answer's default value of 3 and multiplier of 8 both appear verbatim in the evidence.", + "The evidence states the default is 3, which indicates a multiplier of 8, exactly matching the answer.", + "The answer's default value of 3 and multiplier of 8 both appear verbatim in the evidence." + ] + }, + { + "id": "rfc_q02", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states active_connection_id_limit MUST be at least 2, matching the answer's floor claim.", + "The evidence explicitly states the value MUST be at least 2, matching the answer's floor claim.", + "The evidence states the value MUST be at least 2, matching the answer's floor claim exactly." + ] + }, + { + "id": "rfc_q03", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence directly names NO_VIABLE_PATH with code 0x10 for a network path incapable of supporting QUIC.", + "The evidence names NO_VIABLE_PATH with code 0x10, matching the answer exactly.", + "The evidence explicitly names NO_VIABLE_PATH with code 0x10 for this scenario." + ] + }, + { + "id": "rfc_q04", + "requires_crossref": false, + "decomposed": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The answer introduces the identifier \"kInitialRtt\" which never appears in the evidence text.", + "The evidence gives the 333ms value but never names it 'kInitialRtt', so the answer adds an unstated identifier.", + "The evidence states 333 milliseconds but never mentions the identifier kInitialRtt, which the answer adds." + ] + }, + { + "id": "rfc_q05", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states at least one ack-eliciting packet and up to two full-sized datagrams, matching the answer exactly.", + "The evidence states at least one ack-eliciting packet MUST be sent and up to two full-sized datagrams MAY be sent, matching the answer.", + "The evidence supports both the 'at least one' minimum and the 'up to two full-sized datagrams' allowance." + ] + }, + { + "id": "rfc_q06", + "requires_crossref": false, + "decomposed": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives the codepoints 0x30/0x31 but never states that the low bit is the LEN bit, an unsupported added claim.", + "The evidence gives the type values 0x30/0x31 and the bit pattern but never states the low bit is the LEN bit, an added claim.", + "The evidence gives the type values 0x30/0x31 and bit pattern but never states that the low bit is the LEN bit." + ] + }, + { + "id": "rfc_q07", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The version value 0x6b3343cf is stated directly in the evidence and the answer adds no other claims.", + "The evidence states the version 2 field value is 0x6b3343cf, matching the answer with no added claims.", + "The evidence states the version field value 0x6b3343cf exactly as given in the answer." + ] + }, + { + "id": "rfc_q08", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the general Destination Connection ID range as 0-255 bytes versus version 1's 20-byte cap, matching the answer.", + "The evidence states the general range is 0-255 bytes versus version 1's 20-byte cap, matching the answer's comparison.", + "The evidence gives the 0-255 byte general range and the 20-byte version-1 cap, supporting the comparison made." + ] + }, + { + "id": "rfc_q09", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence confirms port 853 as default and explicitly forbids port 53 for DoQ, matching the answer.", + "The evidence states DoQ uses port 853 by default and MUST NOT use port 53, matching the answer.", + "The evidence confirms port 853 as default and explicitly forbids port 53 for DoQ." + ] + }, + { + "id": "rfc_q10", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states at least 1,024 bytes of flow-control credit per unidirectional stream, matching the answer exactly.", + "The evidence states at least 1,024 bytes SHOULD be provided per unidirectional stream, matching the answer exactly.", + "The evidence states the SHOULD provide at least 1,024 bytes of flow-control credit per unidirectional stream." + ] + }, + { + "id": "rfc_q11", + "requires_crossref": false, + "decomposed": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence never gives the identifier 0x01 for the setting nor states that zero capacity forbids all dynamic table insertions.", + "The evidence states the default capacity is zero but never gives the setting identifier 0x01, an added claim.", + "The evidence never states the codepoint 0x01 for SETTINGS_QPACK_MAX_TABLE_CAPACITY, which the answer adds." + ] + }, + { + "id": "rfc_q12", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence lists exactly Content-Length, Content-Type, and Transfer-Encoding as disqualifying header fields.", + "The evidence lists exactly Content-Length, Content-Type, and Transfer-Encoding as disqualifying headers, matching the answer.", + "The evidence lists exactly Content-Length, Content-Type, and Transfer-Encoding as disqualifying headers." + ] + }, + { + "id": "rfc_q13", + "requires_crossref": false, + "decomposed": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the 0-7 range with lower values as higher priority and a default of 3, matching the answer's paraphrase.", + "The evidence states urgency ranges 0-7 with lower values as higher priority and a default of 3, matching the answer.", + "The evidence confirms the 0-7 range, lower-is-higher-priority ordering, and default value of 3." + ] + }, + { + "id": "rfc_q14", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the spin bit is the third-most-significant bit of the first octet, matching the answer exactly.", + "The evidence states the spin bit is the third-most-significant bit of the first octet, matching the answer.", + "The evidence places the spin bit at the third-most-significant bit of the first octet, matching the answer." + ] + }, + { + "id": "rfc_q15", + "requires_crossref": true, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the three-times anti-amplification limit on data sent to an unvalidated address, matching the answer.", + "The evidence states data sent to an unvalidated address is limited to three times data received, matching the answer.", + "The evidence states the three-times anti-amplification limit exactly as claimed in the answer." + ] + }, + { + "id": "rfc_q16", + "requires_crossref": true, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The answer's PTO formula is stated verbatim as the primary formula in the evidence.", + "The evidence gives the exact formula PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay, matching the answer.", + "The answer's formula matches the PTO computation given verbatim in the evidence." + ] + }, + { + "id": "rfc_q17", + "requires_crossref": true, + "decomposed": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence only states the tag is 128 bits; the AEAD_AES_128_GCM computation detail is not present in the evidence.", + "The evidence only states the tag is 128 bits and does not mention AEAD_AES_128_GCM or a Retry pseudo-packet computation, an added claim.", + "The evidence only gives the 128-bit size; it never mentions AEAD_AES_128_GCM or the Retry pseudo-packet computation added by the answer." + ] + }, + { + "id": "rfc_q18", + "requires_crossref": true, + "decomposed": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The salt values match but the evidence never attributes the original salt to \"RFC 9001,\" an added unsupported identifier.", + "The salt values match, but the evidence never attributes the original salt to 'RFC 9001', an added identifier not in the evidence.", + "The evidence names the source as 'the QUIC-TLS document', not 'RFC 9001', which the answer adds as an unstated identifier." + ] + }, + { + "id": "rfc_q19", + "requires_crossref": true, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence directly states the numeric alert description 120 for no_application_protocol.", + "The evidence states the numeric alert description is 120, matching the answer with no added claims.", + "The evidence explicitly gives 120 as the numeric alert description." + ] + }, + { + "id": "rfc_q20", + "requires_crossref": true, + "decomposed": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence describes the requirement for a formal, permanent, public specification beyond expert review, matching the answer.", + "The evidence states Specification Required demands a permanent, public specification beyond expert review, matching the answer.", + "The evidence describes the extra requirement as a stable, public specification beyond expert review, matching the answer." + ] + }, + { + "id": "rfc_q21", + "requires_crossref": true, + "decomposed": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence explicitly states the sub-answers contain no information on this topic, so the answer's specific claims are fabricated.", + "The evidence explicitly states this information is not available in the sub-answers, so the answer's RFC 8441 claim is unsupported.", + "The evidence explicitly states this information is not available, so the answer's specific claims about RFC 8441 and :protocol are unsupported." + ] + }, + { + "id": "rfc_q22", + "requires_crossref": true, + "decomposed": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the ORIGIN frame type is 0xc (12) with zero or more Origin-Entry fields, matching the answer.", + "The evidence states the HTTP/2 ORIGIN frame is type 0xc (12) carrying zero or more Origin-Entry fields, matching the answer.", + "The evidence gives frame type 0xc (12) and zero-or-more Origin-Entry payload contents, matching the answer." + ] + }, + { + "id": "rfc_q23", + "requires_crossref": true, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence directly states the DATAGRAM (0x00) Capsule Type, matching the answer.", + "The evidence states the Capsule Protocol defines the DATAGRAM (0x00) Capsule Type, matching the answer exactly.", + "The evidence explicitly defines the DATAGRAM (0x00) Capsule Type as stated in the answer." + ] + }, + { + "id": "rfc_q24", + "requires_crossref": true, + "decomposed": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence names the parameter and identifier 0x09 but never attributes it to \"RFC 9000,\" an added unsupported identifier.", + "The evidence gives the parameter name and identifier 0x09 but never attributes it to 'RFC 9000', an added identifier not in the evidence.", + "The evidence never attributes the parameter to 'RFC 9000', an identifier the answer adds beyond the stated text." + ] + } + ] +} \ No newline at end of file diff --git a/eval/decisions/synthesis-ab-arm-h-panel.json b/eval/decisions/synthesis-ab-arm-h-panel.json new file mode 100644 index 00000000..b9063b62 --- /dev/null +++ b/eval/decisions/synthesis-ab-arm-h-panel.json @@ -0,0 +1,372 @@ +{ + "arm": "H", + "description": "pooled decomposition, first run (decomposition still sampled at temp 1.0)", + "date": "2026-08-15", + "protocol": "v1 prompts, verifier suffix stripped, 3 blind Sonnet subagent voters, 2-of-3 majority", + "majority_pass": 17, + "n": 24, + "splits": 0, + "note": "superseded by the deterministic H2/G2 comparison in the same decision file", + "rows": [ + { + "id": "rfc_q01", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The answer's default value of 3 and multiplier of 8 both appear verbatim in the evidence.", + "The evidence states the default is 3 and that this indicates a multiplier of 8, matching the answer exactly.", + "Evidence states the default of 3 and the multiplier of 8, matching the answer exactly." + ] + }, + { + "id": "rfc_q02", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly states the value MUST be at least 2, matching the answer's floor claim.", + "The evidence states active_connection_id_limit MUST be at least 2, which is exactly what the answer claims.", + "Evidence states the active_connection_id_limit MUST be at least 2, matching the answer." + ] + }, + { + "id": "rfc_q03", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence names NO_VIABLE_PATH with code 0x10 exactly as stated in the answer.", + "The evidence names NO_VIABLE_PATH with code 0x10 as the error for a network path incapable of supporting QUIC, matching the answer.", + "Evidence names NO_VIABLE_PATH with code 0x10 exactly as the answer states." + ] + }, + { + "id": "rfc_q04", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives 333 milliseconds but never names this value \"kInitialRtt\", so the answer adds an unstated identifier.", + "The evidence gives 333 milliseconds but never names the constant kInitialRtt, so that label is an unsupported addition.", + "The evidence gives the 333 ms value but never names it 'kInitialRtt', an identifier the answer adds without support." + ] + }, + { + "id": "rfc_q05", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states at least one ack-eliciting packet and up to two full-sized datagrams, matching the answer exactly.", + "The evidence states at least one ack-eliciting packet MUST be sent and up to two full-sized datagrams MAY be sent, matching the answer.", + "Evidence states at least one ack-eliciting packet and up to two full-sized datagrams, matching the answer." + ] + }, + { + "id": "rfc_q06", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives the type values 0x30/0x31 and the bit pattern but never states that the low bit is the LEN bit, an added unstated claim.", + "The evidence gives the bit pattern 0b0011000X but never states that the low bit is the LEN bit, so that claim is unsupported.", + "Evidence gives the 0x30/0x31 codepoints but never identifies the low bit as the 'LEN bit', an unsupported added claim." + ] + }, + { + "id": "rfc_q07", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the QUIC v2 version field is 0x6b3343cf, matching the answer.", + "The evidence states the QUIC v2 version field value is 0x6b3343cf, exactly matching the answer.", + "Evidence explicitly states the version 2 field value is 0x6b3343cf, matching the answer." + ] + }, + { + "id": "rfc_q08", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the field is between 0 and 255 bytes generally, versus 20 bytes for version 1, matching the answer.", + "The evidence states the field is between 0 and 255 bytes generally, wider than version 1's 20-byte cap, matching the answer.", + "Evidence states the field is between 0 and 255 bytes and that version 1 caps it at 20 bytes, matching the answer's framing." + ] + }, + { + "id": "rfc_q09", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly states DoQ connections MUST NOT use UDP port 53 and defaults to port 853, matching the answer.", + "The evidence states DoQ uses UDP port 853 by default and MUST NOT use port 53, matching the answer.", + "Evidence states DoQ uses port 853 by default and MUST NOT use port 53, matching the answer." + ] + }, + { + "id": "rfc_q10", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states at least 1,024 bytes of flow-control credit per unidirectional stream, matching the answer exactly.", + "The evidence states at least 1,024 bytes of flow-control credit SHOULD be provided per unidirectional stream, matching the answer exactly.", + "Evidence states at least 1,024 bytes of flow-control credit per unidirectional stream, matching the answer." + ] + }, + { + "id": "rfc_q11", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence never states the setting's identifier 0x01 nor that no dynamic table entries may be inserted, both added claims.", + "The evidence never gives the identifier 0x01 for the setting nor states that no dynamic table entries may be inserted, both of which are added claims.", + "Evidence never gives the setting identifier 0x01 for SETTINGS_QPACK_MAX_TABLE_CAPACITY, an unsupported added detail." + ] + }, + { + "id": "rfc_q12", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence lists exactly Content-Length, Content-Type, and Transfer-Encoding as disqualifying headers, matching the answer.", + "The evidence lists exactly Content-Length, Content-Type, and Transfer-Encoding as disqualifying headers, matching the answer.", + "Evidence lists exactly Content-Length, Content-Type, and Transfer-Encoding as disqualifying headers." + ] + }, + { + "id": "rfc_q13", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states urgency ranges 0 to 7 descending with a default of 3, matching the answer.", + "The evidence states urgency ranges 0 to 7 descending in priority with a default of 3, matching the answer.", + "Evidence states urgency ranges 0 to 7 with default of 3, matching the answer." + ] + }, + { + "id": "rfc_q14", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the spin bit is the third-most-significant bit of byte 0, matching the answer.", + "The evidence states the spin bit is the third-most-significant bit of the first octet of the short header, matching the answer.", + "Evidence states the spin bit is the third-most-significant bit of byte 0 of the short header, matching the answer." + ] + }, + { + "id": "rfc_q15", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states no more than three times the data received, matching the answer exactly.", + "The evidence states a sender MUST limit data to unvalidated addresses to three times the amount received, matching the answer.", + "Evidence states the anti-amplification limit is three times the data received, matching the answer." + ] + }, + { + "id": "rfc_q16", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the PTO formula exactly as smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay, matching the answer.", + "The evidence explicitly gives the formula PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay, matching the answer.", + "Evidence gives the exact formula PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay as stated in the answer." + ] + }, + { + "id": "rfc_q17", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence only states the tag is 128 bits and never mentions AEAD_AES_128_GCM or the Retry pseudo-packet computation, an added claim.", + "The evidence only states the tag is 128 bits and says nothing about it being computed via AEAD_AES_128_GCM, which is an unsupported addition.", + "Evidence only states the tag is 128 bits and does not mention AEAD_AES_128_GCM or a Retry pseudo-packet computation." + ] + }, + { + "id": "rfc_q18", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence explicitly says the original version 1 salt value is not provided in the snippets, contradicting the answer's fabricated salt value.", + "The evidence explicitly states the original version 1 salt value is not provided in the snippets, so the fabricated hex value for it is unsupported.", + "Evidence explicitly says the original version 1 salt value is not provided in the snippets, contradicting the answer's specific hex value." + ] + }, + { + "id": "rfc_q19", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence explicitly states the information could not be found, contradicting the answer's specific value of 120.", + "The evidence explicitly states this information could not be found in the documents, so asserting the value 120 is unsupported.", + "Evidence explicitly states the alert number could not be found, contradicting the answer's claim of 120." + ] + }, + { + "id": "rfc_q20", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence describes the additional requirement as a permanent, readily available public specification, matching the answer.", + "The evidence describes Specification Required as Expert Review plus a stable, permanent, readily available public specification, matching the answer.", + "Evidence states Specification Required is Expert Review plus a formal, permanent, readily available specification, matching the answer." + ] + }, + { + "id": "rfc_q21", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the pseudo-header was defined in RFC 8441 and may appear on request HEADERS for CONNECT, matching the answer.", + "The evidence states the :protocol pseudo-header was first defined for the HTTP/2 CONNECT method extension and may appear on request HEADERS, matching the answer.", + "Evidence states :protocol was defined in RFC 8441 and may appear on request HEADERS for CONNECT, matching the answer." + ] + }, + { + "id": "rfc_q22", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the frame type is 0xc (decimal 12) with zero or more Origin-Entry instances, matching the answer.", + "The evidence states the HTTP/2 ORIGIN frame is type 0xc carrying zero or more Origin-Entry fields, matching the answer.", + "Evidence states the HTTP/2 ORIGIN frame is type 0xc (12) carrying zero or more Origin-Entry fields, matching the answer." + ] + }, + { + "id": "rfc_q23", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the DATAGRAM Capsule Type has value 0x00, matching the answer exactly.", + "The evidence states the Capsule Protocol defines the DATAGRAM (0x00) Capsule Type, matching the answer.", + "Evidence states the DATAGRAM Capsule Type has value 0x00, matching the answer." + ] + }, + { + "id": "rfc_q24", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence never attributes the initial_max_streams_uni parameter to RFC 9000, an added unstated claim.", + "The evidence never states that initial_max_streams_uni is defined in RFC 9000, making that attribution an unsupported addition.", + "Evidence never attributes the 0x09 parameter to RFC 9000, an unsupported identifier added by the answer." + ] + } + ] +} \ No newline at end of file diff --git a/eval/decisions/synthesis-ab-arm-h2-panel.json b/eval/decisions/synthesis-ab-arm-h2-panel.json new file mode 100644 index 00000000..c0d6a790 --- /dev/null +++ b/eval/decisions/synthesis-ab-arm-h2-panel.json @@ -0,0 +1,399 @@ +{ + "arm": "H2", + "description": "pooled decomposition (per-SQ retrieval, single source-aware rerank+synthesis)", + "shared_conditions": "arm G reranker selection + temperature-0 decomposition; identical deterministic row sets (6 decomposed queries, same sub-queries verbatim)", + "date": "2026-08-15", + "protocol": "v1 prompts, verifier suffix stripped, 3 blind Sonnet subagent voters, 2-of-3 majority", + "majority_pass": 17, + "n": 24, + "single_doc_pass": 11, + "crossref_pass": 6, + "decomposed_pass": 3, + "splits": 0, + "rows": [ + { + "id": "rfc_q01", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The default value of 3 and the resulting multiplier of 8 are both stated directly in the evidence.", + "The answer's default value of 3 and multiplier of 8 both appear verbatim in the evidence.", + "Both the default value (3) and the resulting ACK Delay multiplier (8) match the evidence exactly." + ] + }, + { + "id": "rfc_q02", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly states the active_connection_id_limit must be at least 2.", + "The evidence explicitly states the value MUST be at least 2, matching the answer's floor claim.", + "The stated floor of 2 for active_connection_id_limit matches the evidence's MUST-be-at-least-2 requirement." + ] + }, + { + "id": "rfc_q03", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence directly names NO_VIABLE_PATH with code 0x10 for this exact scenario.", + "The evidence directly names NO_VIABLE_PATH with code 0x10, exactly as the answer states.", + "The error code name and value (NO_VIABLE_PATH, 0x10) match the evidence exactly." + ] + }, + { + "id": "rfc_q04", + "requires_crossref": false, + "decomposed": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives 333 milliseconds but never introduces the name 'kInitialRtt', which the answer adds.", + "The evidence never names the constant kInitialRtt, so labeling the 333ms value with that identifier is an unsupported addition.", + "The evidence never names the value 'kInitialRtt'; that identifier is an unsupported addition not present in the evidence." + ] + }, + { + "id": "rfc_q05", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence directly supports both the 'at least one ack-eliciting packet' and 'up to two full-sized datagrams' claims.", + "Both the 'at least one ack-eliciting packet' MUST and the 'up to two full-sized datagrams' MAY are stated in the evidence.", + "The answer restates the evidence's exact figures: at least one ack-eliciting packet, up to two full-sized datagrams." + ] + }, + { + "id": "rfc_q06", + "requires_crossref": false, + "decomposed": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives the 0x30/0x31 codepoints and bit pattern but never states that the low bit is the LEN bit.", + "The evidence gives the type values and bit pattern but never states that the low bit is the LEN bit, an unsupported addition.", + "The evidence gives the type values 0x30/0x31 but never states the low bit is the 'LEN bit', which is an unsupported addition." + ] + }, + { + "id": "rfc_q07", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the QUIC v2 version field value exactly as given in the answer.", + "The evidence states the version 2 field value as 0x6b3343cf, matching the answer exactly.", + "The QUIC version 2 value 0x6b3343cf matches the evidence exactly." + ] + }, + { + "id": "rfc_q08", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence's generic 0-255 byte range and the version 1 20-byte cap directly support the stated comparison.", + "The evidence's unqualified 0-255 byte statement (distinct from the version-1-specific and Version-Negotiation-specific figures) supports the general range claimed.", + "The 0-255 byte range and its comparison to version 1's 20-byte cap both match the evidence." + ] + }, + { + "id": "rfc_q09", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly forbids port 53 and specifies port 853 as the default, matching the answer.", + "The evidence explicitly states DoQ MUST NOT use port 53 and defaults to port 853, matching the answer.", + "The prohibition on port 53 and default use of port 853 both match the evidence exactly." + ] + }, + { + "id": "rfc_q10", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the 1,024 byte minimum flow-control credit verbatim.", + "The evidence states the SHOULD value of at least 1,024 bytes, matching the answer exactly.", + "The 1,024-byte flow-control credit figure matches the evidence exactly." + ] + }, + { + "id": "rfc_q11", + "requires_crossref": false, + "decomposed": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence never gives the '0x01' identifier for SETTINGS_QPACK_MAX_TABLE_CAPACITY or the claim about no entries being insertable, both added by the answer.", + "The evidence never gives the codepoint (0x01) for the setting nor states that no dynamic table entries may be inserted, both unsupported additions.", + "The evidence never gives the setting the identifier 0x01 nor states that no entries may be inserted; both are unsupported additions." + ] + }, + { + "id": "rfc_q12", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence lists exactly these three header fields as disqualifying for the Capsule Protocol.", + "The evidence lists exactly Content-Length, Content-Type, and Transfer-Encoding as the disqualifying headers.", + "The three header fields listed match the evidence's list exactly." + ] + }, + { + "id": "rfc_q13", + "requires_crossref": false, + "decomposed": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence gives the same 0-7 urgency range and default value of 3.", + "The evidence states the urgency range of 0-7 in descending priority order and a default of 3, matching the answer.", + "The 0-7 urgency range and default of 3 both match the evidence exactly." + ] + }, + { + "id": "rfc_q14", + "requires_crossref": false, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the spin bit location exactly as given in the answer.", + "The evidence identifies the spin bit as the third-most-significant bit of the first octet, matching the answer.", + "The bit position (third-most-significant bit of the first octet) matches the evidence exactly." + ] + }, + { + "id": "rfc_q15", + "requires_crossref": true, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the three-times anti-amplification limit exactly as claimed.", + "The evidence states the anti-amplification limit as three times the data received, matching the answer.", + "The three-times-received-data anti-amplification limit matches the evidence exactly." + ] + }, + { + "id": "rfc_q16", + "requires_crossref": true, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The formula in the answer is a direct restatement of the base formula given in the evidence.", + "The answer's formula is stated verbatim as the PTO computation in the evidence.", + "The stated formula is exactly the base PTO formula given in the evidence." + ] + }, + { + "id": "rfc_q17", + "requires_crossref": true, + "decomposed": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence states the tag is 128 bits but never mentions AEAD_AES_128_GCM as the computation mechanism, which the answer adds.", + "The evidence only states the tag is 128 bits and never mentions AEAD_AES_128_GCM or the pseudo-packet computation, an unsupported addition.", + "The evidence states only that the tag is 128 bits; the AEAD_AES_128_GCM computation detail is an unsupported addition." + ] + }, + { + "id": "rfc_q18", + "requires_crossref": true, + "decomposed": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives both salt values but never names the source document as 'RFC 9001', which the answer adds.", + "The evidence attributes the original salt to Section 5.2 of [QUIC-TLS], not to 'RFC 9001', an identifier the answer adds without support.", + "The evidence never names the original salt's source as 'RFC 9001', only as '[QUIC-TLS]'; that citation is an unsupported addition." + ] + }, + { + "id": "rfc_q19", + "requires_crossref": true, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the alert number 120 directly for no_application_protocol.", + "The evidence explicitly gives 120 as the no_application_protocol alert value, matching the answer.", + "The numeric alert value 120 matches the evidence exactly." + ] + }, + { + "id": "rfc_q20", + "requires_crossref": true, + "decomposed": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence explicitly says the policy does not necessarily require 'formal' documentation, contradicting the answer's claim of a formal specification.", + "The evidence explicitly says the policy 'does not necessarily require formal documentation,' contradicting the answer's claim that the specification must be 'formal'.", + "The evidence explicitly notes formal documentation is not necessarily required, so labeling the specification 'formal' is an unsupported addition." + ] + }, + { + "id": "rfc_q21", + "requires_crossref": true, + "decomposed": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence directly supports both the RFC8441 origin and the CONNECT-request context for the :protocol pseudo-header.", + "The evidence confirms :protocol was first defined in RFC 8441 and may appear on CONNECT request HEADERS, matching the answer.", + "The RFC 8441 origin and applicability to CONNECT-method request HEADERS both match the evidence." + ] + }, + { + "id": "rfc_q22", + "requires_crossref": true, + "decomposed": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence gives the frame type 0xc and Origin-Entry payload contents exactly as stated.", + "The evidence gives frame type 0xc (12) carrying zero or more Origin-Entry fields, matching the answer.", + "The frame type 0xc and Origin-Entry payload both match the evidence exactly." + ] + }, + { + "id": "rfc_q23", + "requires_crossref": true, + "decomposed": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence directly names the DATAGRAM (0x00) Capsule Type.", + "The evidence directly states the DATAGRAM (0x00) Capsule Type, matching the answer.", + "The DATAGRAM Capsule Type value 0x00 matches the evidence exactly." + ] + }, + { + "id": "rfc_q24", + "requires_crossref": true, + "decomposed": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives the parameter name and 0x09 identifier but never attributes it to 'RFC 9000', which the answer adds.", + "The evidence never states that initial_max_streams_uni is 'defined in RFC 9000,' an unsupported added identifier.", + "The evidence never states the parameter is 'defined in RFC 9000'; that citation is an unsupported addition." + ] + } + ] +} \ No newline at end of file diff --git a/eval/decisions/synthesis-ab-arm-i-panel.json b/eval/decisions/synthesis-ab-arm-i-panel.json new file mode 100644 index 00000000..81b67989 --- /dev/null +++ b/eval/decisions/synthesis-ab-arm-i-panel.json @@ -0,0 +1,382 @@ +{ + "arm": "I", + "description": "arm H2 + [Source document: <file>] label on every synthesis snippet + prompt rule 7 (attribute via labels)", + "date": "2026-08-15", + "protocol": "v1 prompts, verifier suffix stripped, 3 blind Sonnet subagent voters, 2-of-3 majority", + "majority_pass": 21, + "n": 24, + "single_doc_pass": 12, + "crossref_pass": 9, + "splits": 1, + "vs_arm_h2": { + "gains": [ + "rfc_q06", + "rfc_q18", + "rfc_q20", + "rfc_q24" + ], + "losses": [] + }, + "rows": [ + { + "id": "rfc_q01", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The answer restates the evidence's stated default of 3 and the resulting multiplier of 8 exactly.", + "The evidence states the default is 3 and that this indicates a multiplier of 8, exactly matching the answer.", + "The evidence states the default value 3 for ack_delay_exponent implies a multiplier of 8, matching the answer exactly." + ] + }, + { + "id": "rfc_q02", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly states active_connection_id_limit MUST be at least 2, matching the answer's floor claim.", + "The evidence explicitly states the parameter MUST be at least 2, matching the answer's floor claim.", + "Evidence directly states active_connection_id_limit MUST be at least 2, matching the answer." + ] + }, + { + "id": "rfc_q03", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence directly names NO_VIABLE_PATH with code 0x10 for a path incapable of supporting QUIC, matching the answer.", + "The evidence gives the code NO_VIABLE_PATH (0x10) for a path incapable of supporting QUIC, matching the answer.", + "Evidence names NO_VIABLE_PATH with error code 0x10 as tied to the network path being unable to support QUIC, matching the answer." + ] + }, + { + "id": "rfc_q04", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence gives the 333ms value but never names it kInitialRtt, so that identifier is an unsupported addition.", + "The evidence never uses the name 'kInitialRtt' for the 333ms value, so the answer adds an unstated identifier.", + "The evidence never mentions the name 'kInitialRtt', only the value 333ms, so the added identifier is unsupported." + ] + }, + { + "id": "rfc_q05", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states at least one ack-eliciting packet and up to two full-sized datagrams, matching the answer exactly.", + "The evidence states at least one ack-eliciting packet MUST be sent and up to two full-sized datagrams MAY be sent, matching the answer.", + "Evidence states at least one ack-eliciting packet MUST be sent and up to two full-sized datagrams MAY be sent, matching the answer." + ] + }, + { + "id": "rfc_q06", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence lists the DATAGRAM frame values 0x30/0x31, the 0b0011000X pattern, and the LEN low bit, all matching the answer.", + "The evidence lists the DATAGRAM frame type values 0x30/0x31, the 0b0011000X pattern, and the LEN bit as the low bit, matching the answer exactly.", + "Evidence gives the DATAGRAM frame type values 0x30/0x31 and the LEN bit detail, matching the answer." + ] + }, + { + "id": "rfc_q07", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the QUIC v2 version field value as 0x6b3343cf, matching the answer exactly.", + "The evidence directly states the QUICv2 version field value is 0x6b3343cf, matching the answer.", + "Evidence states the QUICv2 long-header version value is 0x6b3343cf, matching the answer." + ] + }, + { + "id": "rfc_q08", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the DCID field is 0-255 bytes generally versus a 20-byte cap in v1, supporting both parts of the answer.", + "The evidence's general statement of 0-255 bytes (separate from the version-1-specific 20-byte cap) supports the answer's version-independent range claim.", + "Evidence states the general Destination Connection ID field is 0-255 bytes while version 1 caps it at 20 bytes, matching the answer." + ] + }, + { + "id": "rfc_q09", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states DoQ MUST NOT use port 53 and defaults to port 853, matching the answer.", + "The evidence states DoQ MUST NOT use port 53 and defaults to port 853, matching the answer.", + "Evidence states DoQ MUST NOT use port 53 and defaults to port 853, matching the answer." + ] + }, + { + "id": "rfc_q10", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly recommends at least 1,024 bytes of flow-control credit per unidirectional stream, matching the answer.", + "The evidence directly quotes the SHOULD provide at least 1,024 bytes of flow-control credit per unidirectional stream, matching the answer.", + "Evidence's recommendation of at least 1,024 bytes of flow-control credit per unidirectional stream matches the answer." + ] + }, + { + "id": "rfc_q11", + "requires_crossref": false, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence never gives an identifier value for SETTINGS_QPACK_MAX_TABLE_CAPACITY, so the added (0x01) is unsupported.", + "The evidence never states the identifier 0x01 for SETTINGS_QPACK_MAX_TABLE_CAPACITY, so the answer adds an unstated fact.", + "Evidence never gives an identifier '0x01' for SETTINGS_QPACK_MAX_TABLE_CAPACITY, so that added identifier is unsupported." + ] + }, + { + "id": "rfc_q12", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly lists Content-Length, Content-Type, and Transfer-Encoding as disqualifying header fields.", + "The evidence explicitly lists Content-Length, Content-Type, and Transfer-Encoding as disqualifying header fields, matching the answer.", + "Evidence lists Content-Length, Content-Type, and Transfer-Encoding as disqualifying headers, matching the answer exactly." + ] + }, + { + "id": "rfc_q13", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the 0-7 range and default urgency of 3, and the background-task framing supports the descending-priority description.", + "The evidence gives the 0-7 range, default of 3, and identifies 7 as the lowest urgency, supporting the answer's characterization.", + "Evidence states urgency ranges 0-7 with default 3 and labels 7 as the lowest urgency, supporting the descending-order description." + ] + }, + { + "id": "rfc_q14", + "requires_crossref": false, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly places the spin bit at the third-most-significant bit of the first octet, matching the answer.", + "The evidence states the spin bit is the third-most-significant bit of the first octet, matching the answer exactly.", + "Evidence places the spin bit at the third-most-significant bit of the first octet, matching the answer." + ] + }, + { + "id": "rfc_q15", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the anti-amplification limit is three times the data received from the unvalidated address, matching the answer.", + "The evidence states a server MUST limit data sent to an unvalidated address to three times the amount received, matching the answer.", + "Evidence states servers must limit sending to unvalidated addresses to three times received data, matching the answer." + ] + }, + { + "id": "rfc_q16", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence gives the exact PTO formula smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay, matching the answer.", + "The evidence gives the exact formula PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay, matching the answer.", + "Evidence gives the exact PTO formula smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay, matching the answer." + ] + }, + { + "id": "rfc_q17", + "requires_crossref": true, + "votes": [ + false, + false, + false + ], + "majority_pass": false, + "reasons": [ + "The evidence states the tag is 128 bits computed with AEAD_AES_128_GCM but never says it is computed over a 'Retry pseudo-packet', an unsupported addition.", + "The evidence never mentions the tag being computed 'over the Retry pseudo-packet', so the answer adds an unstated detail.", + "Evidence states the tag is computed using AEAD_AES_128_GCM but never specifies it is computed 'over the Retry pseudo-packet', an unsupported addition." + ] + }, + { + "id": "rfc_q18", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence gives both salt values exactly as stated in the answer for v2 and the original v1 salt.", + "The evidence gives both salt values exactly as quoted and attributes them correctly to version 2 and version 1, matching the answer.", + "Evidence gives both the v2 and v1 salts exactly as stated in the answer." + ] + }, + { + "id": "rfc_q19", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the no_application_protocol alert value is 120, matching the answer.", + "The evidence states no_application_protocol is alert value 120, matching the answer.", + "Evidence states the no_application_protocol alert value is 120, matching the answer." + ] + }, + { + "id": "rfc_q20", + "requires_crossref": true, + "votes": [ + true, + false, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence describes Specification Required as Expert Review plus a permanent, readily available public specification, matching the answer.", + "The evidence explicitly says the specification can be informal documentation published outside the RFC path, contradicting the answer's claim that it must be 'formal'.", + "Evidence describes Specification Required as Expert Review plus a permanent, readily available public specification, matching the answer's core claim." + ] + }, + { + "id": "rfc_q21", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states :protocol was defined in RFC 8441 and may be included on request HEADERS for the CONNECT tunnel, matching the answer.", + "The evidence states :protocol was defined in RFC8441 and MAY be included on request HEADERS for the CONNECT tunnel, matching the answer.", + "Evidence states :protocol was defined in RFC8441 and may appear on CONNECT request HEADERS, matching the answer." + ] + }, + { + "id": "rfc_q22", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence states the HTTP/2 ORIGIN frame type is 0xc (12) with zero or more Origin-Entry fields, matching the answer.", + "The evidence states the HTTP/2 ORIGIN frame is type 0xc (12) carrying zero or more Origin-Entry fields, matching the answer.", + "Evidence states the HTTP/2 ORIGIN frame type is 0xc (12) carrying zero or more Origin-Entry fields, matching the answer." + ] + }, + { + "id": "rfc_q23", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence explicitly defines the DATAGRAM (0x00) Capsule Type, matching the answer.", + "The evidence defines the DATAGRAM (0x00) Capsule Type, matching the answer.", + "Evidence defines the DATAGRAM Capsule Type as value 0x00, matching the answer." + ] + }, + { + "id": "rfc_q24", + "requires_crossref": true, + "votes": [ + true, + true, + true + ], + "majority_pass": true, + "reasons": [ + "The evidence identifies initial_max_streams_uni as parameter 0x09 and the three-stream HTTP/3 requirement, matching the answer.", + "The evidence gives initial_max_streams_uni with identifier 0x09 and the requirement to allow at least three unidirectional streams, matching the answer.", + "Evidence identifies initial_max_streams_uni as transport parameter 0x09 tied to the three-unidirectional-stream requirement, matching the answer." + ] + } + ] +} \ No newline at end of file diff --git a/eval/decisions/synthesis-ab-sonnet-panels.json b/eval/decisions/synthesis-ab-sonnet-panels.json new file mode 100644 index 00000000..347693fa --- /dev/null +++ b/eval/decisions/synthesis-ab-sonnet-panels.json @@ -0,0 +1,922 @@ +{ + "b": { + "pass": 6, + "sd": [ + 3, + 14 + ], + "xr": [ + 3, + 10 + ], + "splits": 3, + "per_row": { + "rfc_q01": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q02": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q03": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q04": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q05": { + "votes": [ + false, + false, + true + ], + "maj": false, + "xr": false + }, + "rfc_q06": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q07": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": false + }, + "rfc_q08": { + "votes": [ + false, + false, + true + ], + "maj": false, + "xr": false + }, + "rfc_q09": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": false + }, + "rfc_q10": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q11": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q12": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": false + }, + "rfc_q13": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q14": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q15": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q16": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q17": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q18": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q19": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q20": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q21": { + "votes": [ + false, + false, + true + ], + "maj": false, + "xr": true + }, + "rfc_q22": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q23": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q24": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + } + } + }, + "c": { + "pass": 7, + "sd": [ + 2, + 14 + ], + "xr": [ + 5, + 10 + ], + "splits": 0, + "per_row": { + "rfc_q01": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q02": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q03": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q04": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q05": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q06": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": false + }, + "rfc_q07": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": false + }, + "rfc_q08": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q09": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q10": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q11": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q12": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q13": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q14": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q15": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q16": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q17": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q18": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q19": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q20": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q21": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q22": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q23": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q24": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + } + } + }, + "d": { + "pass": 4, + "sd": [ + 2, + 14 + ], + "xr": [ + 2, + 10 + ], + "splits": 2, + "per_row": { + "rfc_q01": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": false + }, + "rfc_q02": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q03": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q04": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q05": { + "votes": [ + false, + false, + true + ], + "maj": false, + "xr": false + }, + "rfc_q06": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q07": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": false + }, + "rfc_q08": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q09": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q10": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q11": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q12": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q13": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q14": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q15": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q16": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q17": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q18": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q19": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q20": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q21": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q22": { + "votes": [ + false, + false, + true + ], + "maj": false, + "xr": true + }, + "rfc_q23": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q24": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + } + } + }, + "d2": { + "pass": 5, + "sd": [ + 1, + 14 + ], + "xr": [ + 4, + 10 + ], + "splits": 2, + "per_row": { + "rfc_q01": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q02": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q03": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q04": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q05": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q06": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q07": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": false + }, + "rfc_q08": { + "votes": [ + true, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q09": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q10": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q11": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q12": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q13": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q14": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": false + }, + "rfc_q15": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q16": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q17": { + "votes": [ + true, + false, + true + ], + "maj": true, + "xr": true + }, + "rfc_q18": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q19": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q20": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + }, + "rfc_q21": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q22": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q23": { + "votes": [ + true, + true, + true + ], + "maj": true, + "xr": true + }, + "rfc_q24": { + "votes": [ + false, + false, + false + ], + "maj": false, + "xr": true + } + } + } +} \ No newline at end of file diff --git a/eval/decisions/synthesis-grounding-ab-2026-08-13.md b/eval/decisions/synthesis-grounding-ab-2026-08-13.md new file mode 100644 index 00000000..4807a4c3 --- /dev/null +++ b/eval/decisions/synthesis-grounding-ab-2026-08-13.md @@ -0,0 +1,272 @@ +# Synthesis grounding A/B (improvement_plan 1.9) โ€” 2026-08-13 + +Question: can prompt, sampling, or a bigger local model fix the measured +synthesis-grounding failure on unseen dense technical text (RFC corpus +baseline: 5/24 answers contain the gold fact despite recall@20 0.958)? + +Five arms on the identical 24-question rfc gold set and post-fix index; all +answers judged by the validated Sonnet subagent panel (3 voters per arm, +blind, isolation-instructed; per-row votes in +`synthesis-ab-sonnet-panels.json`). Arms generated by an Opus agent +(harness monkeypatches only; no repo edits); panels and verdicts by the gate. + +| arm | prompt | model | sampling | pass | single-doc | crossref | judge splits | +|---|---|---|---|---|---|---|---| +| A (baseline) | shipped | qwen3.5:9b | default | 5/24 | 1/14 | 4/10 | โ€” | +| B | strict | qwen3.5:9b | default | 6/24 | 3/14 | 3/10 | 3 | +| **C** | **strict** | **qwen3.5:9b** | **temp 0** | **7/24** | 2/14 | 5/10 | **0** | +| D | strict | qwen3.6:35b-a3b | default | 4/24 | 2/14 | 2/10 | 2 | +| D2 | shipped | qwen3.6:35b-a3b | default | 5/24 | 1/14 | 4/10 | 2 | + +The strict prompt (now shipped) removes the old prompt's clearly-labelled +"General knowledge" escape hatch and forbids quotes, section numbers and +document names not present in the snippets. Temperature 0 applies to the +synthesis stream and the agent's compose stream (arm C's exact tested +config; composer *prompt* remains unmodified โ€” see the compose gap below). + +## Verdicts + +* **Bigger model โ€” REJECTED.** qwen3.6:35b-a3b (35B MoE, 3B active) scores + equal-or-worse than the 9b under both prompts (4/24, 5/24). It is ~2.1ร— + faster per query on this machine (39.7s vs 83.5s, shared-GPU caveat) and + produced the only degenerate output of the experiment (a 69,986-char + runaway on `rfc_q20`). Grounding on dense unseen text is not a parameter- + count problem at this scale. gemma4:31b untested. +* **Arm C โ€” ADOPTED as default, stated honestly:** the +2 quality delta + (5โ†’7) is at this harness's noise floor and is NOT claimed as a measured + improvement. Adoption rests on: (a) both strict-prompt 9b arms โ‰ฅ baseline + (direction consistent); (b) escape-hatch removal is what the observed + fabricated-citation failures exploited; (c) temperature 0 produced the + only arm with zero judge splits and directly targets the documented + temp-1.0 smoke/eval flake; (d) zero cost. +* **Item 1.9 remains OPEN.** Best measured config answers 7/24 on this + corpus. The failure is model-behavior (prior-driven answers survive even + under the strict prompt: "Based on standard QUIC specificationsโ€ฆ"), and + no local model tested fixes it. + +## Mechanistic finding: the compose gap + +Query decomposition fired on 24/24 queries; when it yields >1 sub-query the +judged answer is written by the agent's *composer* (`loop.py`), whose prompt +carries none of the strict grounding rules โ€” 4-9 of 24 rows per arm. The +strict prompt therefore never governed a third of the judged outputs. +Extending the grounding rules to the compose prompt is the highest-leverage +untested follow-up, ahead of abstain-on-low-verifier-confidence and +passage-level citation forcing. + +## Also recorded + +* Zero refusal-string answers in any arm (the mandated abstain sentence was + never emitted, in 120 answers) โ€” the abstain path may be dead in practice. +* Retrieval identical across arms (cited-expected-source 24/24, one 23/24), + so all deltas are synthesis-side. +* `stream_completion` gained an `options` kwarg (temperature verified + reaching the Ollama payload at the wire). + +--- + +## Arm E โ€” strict compose prompt: TESTED AND REJECTED (2026-08-13) + +The compose-gap hypothesis was tested as a single code change on top of the +shipped arm C: the composer prompt in `loop.py` was given the same hard +grounding rules as the synthesis prompt (plus removal of its โ‰ค5-sentence +cap), everything else unchanged. 24 fresh E2E answers, judged by the same +3-voter Sonnet panel (per-row votes in `synthesis-ab-arm-e-panel.json`). + +**Result: 4/24 (single-doc 1/14, crossref 3/10) vs arm C's 7/24 โ€” a +regression, and an attributable one.** Compose fired on 8/24 rows this run; +all three rows that flipped passโ†’fail versus arm C (`rfc_q20`, `rfc_q21`, +`rfc_q22`, all crossref) were composed rows, while the one gained row and +the one unrelated loss never touched the composer. Zero gains among the 8 +composed rows. The mandated abstain sentence also fired for the first time +in the whole experiment (1/24) โ€” the strict rules do change composer +behavior; they change it for the worse. + +**Why, best reading:** the composer's input is already-synthesized +sub-answer prose, not raw snippets. "Copy character-for-character and write +nothing not present" is the right contract against raw evidence, but +against loosely-phrased intermediate prose it makes the composer drop or +refuse facts the sub-answers actually carried. The old composer prompt +(softer "use only the sub-answers" + its 5-sentence cap) surfaces the key +fact more reliably. + +The change is reverted; the shipped configuration remains exactly arm C. +Caveat: decomposition still samples at temperature 1.0, so the composed-row +*set* differs between runs โ€” the 8-row attribution is clean, but a rerun +would compose different rows. Remaining 1.9 levers, in order: +abstain-on-low-verifier-confidence, deterministic decomposition +(temperature 0 there too, which would also stabilize these comparisons), +passage-level citation forcing. + +--- + +## Arm F โ€” cross-leg dedupe + 12k synthesis context budget: ADOPTED (2026-08-14) + +Root cause finally measured, not guessed: every synthesis call was built +from ~94k tokens of context ("512-token" chunks store at ~823 tokens after +enrichment prefixes; two retrieval legs returned 20+20 with no dedupe; the +ยฑ1 sibling merge tripled each entry), while Ollama's parallel-slot split +served only ~16k of window and silently FRONT-truncated โ€” the model saw the +tail of the ranking, i.e. the *worst* retrieved evidence, on essentially +every call. + +Change under test (on top of shipped arm C, everything else identical): + +1. **Cross-leg dedupe** by `(document_id, chunk_index)` at the retrieval + union point in `retrieval_pipeline.py` (improvement_plan 1.5). +2. **`_budget_synthesis_context()`**: rank-ordered packing into an explicit + token budget (default 12,000; config-overridable via + `synthesis_context_tokens`), sibling-span overlap suppression for + latechunk-merged entries, minimum one doc always kept. +3. **Slot-proof truncation warning** in `ollama_client.py` + (`prompt_eval_count < prompt_chars // 6` catches the served-window split + the old num_ctx comparison was blind to). +4. `.env.example`: `OLLAMA_NUM_PARALLEL=1` note. + +Mechanics, verified from the run log (32 synthesis calls across 24 queries): +dedupe 40 โ†’ 30โ€“39 per query; budget kept top 4โ€“5 merged docs per call; +context mean 40,367 chars โ‰ˆ 11.5k tokens (max 41,955) vs ~335k chars +before; **FRONT-TRUNCATED warnings: 0** (previously routine). Runtime +1,595s โ†’ 1,080s total (โˆ’32%), median 59s โ†’ 45s per query. + +**Result: 16/24 (single-doc 10/14, crossref 6/10) vs arm C's 7/24 +(2/14, 5/10).** Per-row votes in `synthesis-ab-arm-f-panel.json`. Panel +near-unanimous: 1 split across 72 votes (rfc_q21, 2โ€“1 pass). Voter totals +15/16/16. Mechanical checks equal-or-better: cited-expected-source 24/24 +(arm C 23/24), answer-contains-expected 24/24 both. + +The single-doc jump (2/14 โ†’ 10/14) is exactly where truncation hurt most: +those questions had the right chunk ranked #1, and #1 was the first thing +the front-truncation deleted. Crossref moved less (5/10 โ†’ 6/10) โ€” +multi-hop composition losses are a different failure mode (decomposition +at temp 1.0, composer contract), still on the 1.9 backlog. + ++9/24 with a near-unanimous panel is far outside the established 1โ€“2 row +noise floor. **Adopted; committed.** + +--- + +## Arm G โ€” final-stage reranker with threshold selection: ADOPTED (2026-08-15) + +User-directed design: with the budget now controlling *how much* context +synthesis gets, use a reranker to control *which* and *how many* docs get in +โ€” relevance-threshold **selection**, not just reordering, so easy questions +send a small clean context instead of a fixed-size one. + +Change on top of arm F (all in-tree): + +1. `reranker.enabled: true` by default with **Qwen3-Reranker-4B** (the + calibrated yes/no-logit scorer โ€” its P(relevant) makes a threshold + meaningful; raw cross-encoder logits would not). +2. New `reranker.min_score: 0.5` โ€” union-of-max semantics: a candidate is + kept if its best score against ANY query (original + sub-queries when + present) clears the bar. `min_keep: 3` floors the selection; + `top_k: 10` caps it; the 12k budget remains the backstop. + Threshold applies only when the scorer is the calibrated Qwen class. +3. Candidates are scored on their **core chunk text** (preserved in + `metadata.core_text` at merge time), not the ยฑ1-merged block โ€” the merge + buries the matching chunk mid-string past the scorer's 2,048-token + truncation window and dilutes its signal. + +Mechanics (32 synthesis calls): selection kept mean 8.8 docs (range 3โ€“10 โ€” +genuinely adaptive; one query kept 3/36); context mean 37,196 chars (arm F +40,367); truncation 0; cited-expected-source 24/24. Cost: median query +45s โ†’ 71s, total 1,080s โ†’ 2,075s (+92%) โ€” the 4B scorer costs ~25โ€“30s per +rerank pass on MPS. + +**Result: 18/24 (single-doc 11/14, crossref 7/10) vs arm F's 16/24 +(10/14, 6/10).** Gains rfc_q02/q17/q22, loss rfc_q20 (2โ€“1 split; answer +substantively correct on "permanent, readily available public +specification" but hedged formality โ€” judge nuance, not a selection +failure). One split across 72 votes. + +**Honest framing: net +2 is within the established 1โ€“2 row noise floor โ€” +adopted NOT as a claimed quality win but as: user-directed feature, zero +quality regression with equal-or-better subsets on both categories, +verified adaptive-selection behavior, at a real and disclosed 2ร— latency +cost.** Escape hatches: `reranker.enabled: false` restores arm F; +`RERANKER_MODEL=Qwen/Qwen3-Reranker-0.6B` would cut latency (quality +unmeasured). The Phase-1 "reranker off by default" call is superseded: it +predates the context budget, when rank order barely mattered because +front-truncation discarded the top of the list anyway. + +--- + +## Arms H/H2/G2 โ€” pooled decomposition + deterministic decomposition: ADOPTED (2026-08-15) + +**The architecture change (arm H, user-directed):** decomposed queries used +to run one full pipeline per sub-query (N rerank passes, N synthesis calls) +and compose the sub-ANSWERS. Now they run per-sub-query *retrieval only*, +pool + dedupe the candidates (`_pooled_first_stage`, tagging each with its +source sub-queries), run ONE rerank pass โ€” each candidate scored only +against the sub-queries that retrieved it (union-of-max, same pair cost as +the old per-SQ passes) with a per-sub-query floor so no sub-question's +evidence can be entirely thresholded out โ€” and ONE synthesis against the +original question. The composer drops out of this path entirely. +Config: `query_decomposition.{compose_from_sub_answers: false, +pooled_first_stage: true}`. + +**First measurement (arm H vs G, sampled decomposition): 17/24 vs 18/24** +โ€” but only one loss was attributable to pooling (rfc_q17, an unsupported +detail the per-SQ context happened to contain); the other flips were on +the code-identical direct path, i.e. temp-1.0 decomposition noise. That +noise had polluted three arms of comparisons, so per user decision the +noise source was fixed first and both arms re-run. + +**Deterministic decomposition:** `QueryDecomposer` now decodes greedily +(`options={"temperature": 0}` via a new `options` kwarg on +`generate_completion`, wire-verified). Probe: identical splits and +identical sub-query text across repeat runs. Bonus effect: all 18 +direct-path rows received IDENTICAL panel verdicts across the two re-run +arms โ€” the first A/B in this project with zero direct-row judge noise. +(Answers are not byte-identical across arms โ€” the num_ctx ratchet shifts +decode numerics โ€” but verdicts were.) + +**Clean re-run (H2 = pooled, G2 = composer forced via +`compose_sub_answers=True`, identical deterministic row sets, 6 decomposed +queries with verbatim-identical sub-queries): 17/24 vs 17/24. Dead tie.** +Identical subsets (single-doc 11/14, crossref 6/10 both), zero split votes +across all 144 judgments, decomposed rows 3/6 each (one borderline flip in +each direction: q20 G-only, q21 H-only). Wall time tied at N=2 +sub-queries (1,494s vs 1,500s). + +**Verdict: quality is a true tie; structure decides. ADOPTED pooled + +temp-0 decomposition:** 24 synthesis calls instead of 32 (scales linearly, +not multiplicatively, with sub-query count), the composer โ€” where arm E +showed facts get lost โ€” is gone from the decomposition path, and A/B row +sets are stable run-to-run from here on. Panel records: +`synthesis-ab-arm-h-panel.json` (first run), `synthesis-ab-arm-h2-panel.json`, +`synthesis-ab-arm-g2-panel.json`. + +--- + +## Arm I โ€” source-document labels in the synthesis context: ADOPTED (2026-08-15) + +Diagnosis of the four crossref failures surviving arm H2 showed two were +**attribution failures, not fact failures**: rfc_q18 had both Initial-keys +salts character-for-character correct and rfc_q24 had the parameter name +and 0x09 identifier correct โ€” both failed only for not naming the source +RFC. Root cause was ours: the synthesis context was a bare join of chunk +texts with no source labels, while strict-prompt rule 3 (correctly) +forbids writing document names not present in the snippets. The corpus +cross-references by tag ("[QUIC-TLS]"), so the model *could not* say +"RFC 9001" without breaking its grounding contract โ€” even though the +pipeline knows every chunk's source file. + +Change: every snippet in the synthesis context now opens with +"[Source document: <document_id>]", and prompt rule 7 tells the model to +attribute facts via those labels when sourcing matters. Grounding stays +strict โ€” document names are now *in* the snippets. + +**Result: 21/24 (single-doc 12/14, crossref 9/10) vs arm H2's 17/24 +(11/14, 6/10). Gains rfc_q06/q18/q20/q24, zero losses, one split +(rfc_q20, 2โ€“1 pass).** Identical row set to H2 (deterministic +decomposition, 6 decomposed queries). +4 with no regressions is well +outside the 1โ€“2 row noise floor: the first clearly-attributable quality +win at synthesis since the context-budget fix. rfc_q17 also now includes +the AEAD_AES_128_GCM computation (retrieval drew the ยง5.8 chunk this +run) but stays failed on a nuance; rfc_q10 and rfc_q15 remain the +single-doc residue. + +RFC-corpus arc: 5 โ†’ 7 (strict prompt) โ†’ 16 (dedupe+budget) โ†’ 17โ€“18 +(reranker selection / pooled, tie) โ†’ **21 (source labels)**. diff --git a/eval/decisions/union-fusion-2026-08-20.md b/eval/decisions/union-fusion-2026-08-20.md new file mode 100644 index 00000000..1f2bbfa3 --- /dev/null +++ b/eval/decisions/union-fusion-2026-08-20.md @@ -0,0 +1,45 @@ +# Candidate-pool union fusion (FTS โˆช dense โˆช MV โ†’ reranker) โ€” workload-dependent, not a default + +**Date:** 2026-08-20 ยท **Judge:** Sonnet throughout ยท **Follow-up to:** paraphrase-robustness-2026-08-20.md + +## What was tested + +The fusion recommendation from the paraphrase study, implemented as `MV_UNION=1` +(retrievers.py): run all three legs (FTS, dense, LFM2.5-ColBERT MaxSim) at k=20 each, +skip the RRF top-k cut, and hand the FULL union (~47 unique candidates measured on rfc) +to the Qwen3-Reranker-4B to arbitrate. A disagreeing leg can then only ADD candidates โ€” +never push another leg's find out of the pool. Rerank pool ~2.3x โ†’ per-query latency +roughly +30โ€“40%. + +## Results (Sonnet bulk + 3-voter panels on every flip) + +| question set | control (2-leg hybrid) | 3-leg RRF | 3-leg UNION | +|---|---|---|---| +| paraphrased | 95 | 97 (net +2, split votes) | **99 โ€” panel net +4 REAL** (6 gains / 2 losses, 24/24 votes unanimous) | +| original | 100 | 100 (net โˆ’1) | 95 โ€” panel net **โˆ’3 REAL** (docs_d05/d07/d22, unanimous; 4 other flips dissolved as judge noise) | + +## Findings + +1. **The fusion diagnosis was correct.** Same legs as 3-leg RRF, only the fusion + changed, and the paraphrase-set result went from +2 (with split votes) to +4 + (unanimous) โ€” the largest verified gain of the whole multi-vector investigation. + Gains include four rows paraphrasing had broken (docs_d08/d13/d19, rfc_q19). +2. **Union amplifies in both directions.** On original questions the wider pool admits + distractors: all three real losses are docs โ€” 608 chunks of near-duplicate + documentation text โ€” where the reranker, shown 47 candidates instead of 20, + sometimes prefers a plausible-but-wrong chunk the RRF cut used to hide from it. +3. Net across both sets: +1. Not a default. + +## Decision + +Defaults unchanged (2-leg hybrid, RRF). The measured configuration guide: + +- **Document-phrased queries** (users quote the docs' vocabulary): shipped 2-leg hybrid. Best cell: 100. +- **Paraphrase-heavy / conversational queries**: `MV_RETRIEVAL_ENDPOINT` + `MV_UNION=1`. Best cell: 99 (+4 real over 2-leg), at ~+30โ€“40% latency and the MV sidecar/storage cost. + +## Untested refinements (recorded, not run) + +- Cap the union's per-leg contribution (top-10 per leg instead of top-20) to shrink the + distractor surface on dense corpora. +- Gate the MV leg on FTS-confidence (add MV candidates only when the BM25 top score is + weak โ€” a proxy for "the query doesn't match document wording"). diff --git a/eval/finalize_goldset.py b/eval/finalize_goldset.py new file mode 100644 index 00000000..d44d6215 --- /dev/null +++ b/eval/finalize_goldset.py @@ -0,0 +1,169 @@ +"""Apply the human verification pass to the raw generated queries. + +Every one of the 72 generated (query, anchor) pairs was read against its source +document by hand. The outcome of that pass is recorded here, per row, so the +committed gold set is auditable rather than "trust me": + + accepted the model's question was answerable from the anchor, unambiguous, + and did not simply restate the answer โ€” used verbatim. + rescued the model ignored the JSON key and returned {"question": ...} or a + truncated string; the question was hand-written from the anchor + (the raw model output is kept in the row for audit). + rewritten the question was wrong, vague, or leaked the whole expected string + into the query (which would hand the lexical leg a free win). + discarded not answerable from the source; dropped entirely. + + .venv/bin/python eval/finalize_goldset.py + +Reads eval/goldset/_generated/<corpus>.raw.jsonl, writes eval/goldset/<corpus>.jsonl. +""" + +import json +import os +import sys + +EVAL_DIR = os.path.dirname(os.path.abspath(__file__)) +RAW_DIR = os.path.join(EVAL_DIR, "goldset", "_generated") +OUT_DIR = os.path.join(EVAL_DIR, "goldset") +CORPORA_DIR = os.path.join(EVAL_DIR, "corpora") + +SIDECARS = { + "atlas7": "atlas7_service_manual.facts.json", + "hr": "northwind_leave_policy.facts.json", + "docs": "repo_docs.facts.json", +} + +# id -> (verdict, replacement query or None, replacement fact_ids or None, reason) +EDITS = { + # ---- Atlas-7 ----------------------------------------------------------- + "atlas7_a02": ("rewritten", "What pressure is the steam boiler held at?", None, + "model wrapped its output in literal quote characters"), + "atlas7_a03": ("rescued", "What temperature does the PID hold the brew water at?", None, + "model returned {\"question\": ...}; question taken from that payload"), + "atlas7_a05": ("rewritten", "What is the water hardness threshold for this machine?", None, + "'What is the hardness threshold?' had no anchor to the document at all"), + "atlas7_a08": ("rewritten", "Who makes the Atlas-7, and where are they based?", None, + "generated question asked who manufactures the manufacturer"), + "atlas7_a11": ("rescued", "How tightly is the brew water temperature controlled?", None, + "model returned {\"question\": ...}"), + "atlas7_a16": ("rewritten", "The machine is not registering any water flow at all - what should I check?", None, + "'stops delivering water' was ambiguous between the E42 and E57 procedures"), + "atlas7_a19": ("rewritten", "Which needs doing more often on this machine: descaling, or replacing the group head gasket?", None, + "generated comparative leaked the descaling interval into the question"), + "atlas7_a20": ("rescued", "Which error code points to a failed temperature sensor, and which one points to excess steam pressure?", None, + "model returned a truncated {\"question\": ...} payload"), + "atlas7_a21": ("rewritten", "What kind of descaling product will void the warranty?", None, + "generated question contained the expected answer text verbatim"), + "atlas7_a22": ("rewritten", "Will a warranty claim be accepted without a serial number, and where do I find it?", + ["atlas_serial_location"], + "generated question restated '120 ppm'; re-anchored to the serial-number requirement"), + "atlas7_a23": ("rewritten", "Is there a limit on how far the brew temperature may drift before it is out of spec?", None, + "reworded so it is not a near-duplicate of a11's phrasing"), + + # ---- HR handbook ------------------------------------------------------- + "hr_h08": ("rewritten", "What is the identifier and revision number of the leave policy?", None, + "generated question quoted 'PPL-204 revision 4', i.e. the expected string"), + "hr_h09": ("rewritten", "Which department owns this policy, and where is it based?", None, + "generated question asked who owns the department, inverting the fact"), + "hr_h12": ("rewritten", "After the initial full-pay period of a long illness ends, what proportion of salary continues and for how long?", None, + "generated question invented a 'one month / next two months' schedule the policy does not state"), + "hr_h14": ("rewritten", "How far in advance do I have to file a leave request, and where do I file it?", None, + "generated question contained 'Kestrel HR portal', part of the expected string"), + "hr_h15": ("rewritten", "At what point during a sickness absence do I have to produce a doctor's note?", None, + "generated question contained '4 consecutive working days', the expected string"), + "hr_h16": ("rewritten", "What do I need to qualify for an extended unpaid break, how much notice must I give, and who signs it off?", None, + "'a leave away from work' was too vague to be answerable by the sabbatical section specifically"), + "hr_h18": ("rescued", "How does annual leave entitlement differ between employees below Grade 7 and those at Grade 7 or above?", None, + "model returned a truncated {\"question\": ...} payload"), + "hr_h19": ("rewritten", "How does sick pay in the first three months of an absence compare with the months that follow?", None, + "generated question referenced 'after 20 weeks', which is past the end of the stated schedule"), + + # ---- Repo documentation ------------------------------------------------ + "docs_d07": ("rewritten", "How many extra vectors does turning on late chunking write?", None, + "generated question compared against 'early chunking', a term the docs never use"), + "docs_d08": ("rewritten", "What does the enricher do when the model returns an almost-empty summary?", None, + "generated question restated 'shorter than 5 characters'"), + "docs_d09": ("rewritten", "How large is the per-retriever cache that stores previously embedded queries?", None, + "generated question invented 'for each search engine'"), + "docs_d10": ("rewritten", "How many model calls does knowledge-graph extraction spend on each chunk?", None, + "'each processed unit of information' was too vague to be answerable"), + "docs_d11": ("rewritten", "What confidence value is interpreted as a failed parse rather than a real score?", None, + "tightened so the answer is the value, not the behaviour"), + "docs_d14": ("rescued", "Why do documents indexed from the command line end up with a different chunk size than ones indexed through the HTTP API?", None, + "model returned a truncated, off-topic {\"question\": ...} payload"), + "docs_d15": ("rewritten", "How are plain-text uploads processed differently from PDFs and Word files?", None, + "generated question presupposed a 'standard processing pipeline' the docs do not name"), + "docs_d16": ("rewritten", "Does this project build an approximate-nearest-neighbour index, and what does that mean for how a vector query executes?", None, + "generated question contained both expected strings"), + "docs_d17": ("rewritten", "Which model handles routing and verification, and how much extra work does verification add per query?", None, + "generated question was about swapping models, which the anchors do not cover"), + "docs_d18": ("rewritten", "Does the indexer overlap chunks or parallelise the work across workers?", None, + "generated question asked about 'distributed systems features', not answerable from the anchors"), + "docs_d19": ("rewritten", "Can I tune how much the keyword leg counts versus the vector leg when the two are combined?", None, + "generated question contained the expected string 'weighted linear blend'"), + "docs_d20": ("rewritten", "Does the generated answer contain markers pointing at the passage each claim came from?", None, + "generated question contained the expected string 'inline citation marker'"), + "docs_d21": ("rewritten", "Does the agent do any pattern matching on the query before it asks a model to route it?", None, + "generated question contained the expected string 'regex or keyword stage'"), + "docs_d24": ("rescued", "Why does a vector-dimension mismatch raise an error instead of just rebuilding the table?", None, + "model returned {\"question\": ...}"), +} + + +def load_facts(corpus: str) -> dict: + with open(os.path.join(CORPORA_DIR, SIDECARS[corpus]), "r", encoding="utf-8") as fh: + return {f["id"]: f for f in json.load(fh)["facts"]} + + +def main() -> int: + tally = {"accepted": 0, "rescued": 0, "rewritten": 0, "discarded": 0} + for corpus in sorted(SIDECARS): + facts = load_facts(corpus) + raw_path = os.path.join(RAW_DIR, f"{corpus}.raw.jsonl") + if not os.path.exists(raw_path): + print(f"missing {raw_path}; run eval/build_goldset.py first") + return 1 + + out_rows = [] + with open(raw_path, "r", encoding="utf-8") as fh: + for line in fh: + row = json.loads(line) + verdict, new_query, new_fact_ids, reason = EDITS.get( + row["id"], ("accepted", None, None, None)) + tally[verdict] += 1 + if verdict == "discarded": + continue + + fact_ids = new_fact_ids or row["fact_ids"] + anchors = [facts[fid] for fid in fact_ids] + out_rows.append({ + "id": row["id"], + "corpus": corpus, + "query": new_query or row["query"], + "expected": [a["expected"] for a in anchors], + "match": row["match"], + "fact_ids": fact_ids, + "answer": " ".join(a["summary"] for a in anchors), + "dimensions": {**row["dimensions"], "topic": anchors[0]["topic"]}, + "verification": { + "verdict": verdict, + "reason": reason, + "generated_query": row["query"] or row.get("raw_response"), + "generator_model": row["generator_model"], + }, + }) + + out_path = os.path.join(OUT_DIR, f"{corpus}.jsonl") + with open(out_path, "w", encoding="utf-8") as fh: + for row in sorted(out_rows, key=lambda r: r["id"]): + fh.write(json.dumps(row, ensure_ascii=False) + "\n") + print(f"{corpus}: wrote {len(out_rows)} rows -> {out_path}") + + total = sum(tally.values()) + print(f"\nverification pass over {total} generated pairs: " + + ", ".join(f"{k}={v}" for k, v in tally.items())) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/goldset/_generated/atlas7.raw.jsonl b/eval/goldset/_generated/atlas7.raw.jsonl new file mode 100644 index 00000000..5d3b1018 --- /dev/null +++ b/eval/goldset/_generated/atlas7.raw.jsonl @@ -0,0 +1,24 @@ +{"id": "atlas7_a01", "corpus": "atlas7", "query": "What is the pressure of the brew boiler during extraction?", "expected": ["pressure of 9.2 bar"], "match": "any", "fact_ids": ["atlas_brew_pressure"], "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a02", "corpus": "atlas7", "query": "\"What is the pressure at which the steam boiler is maintained?\"", "expected": ["steam boiler is maintained at 1.45 bar"], "match": "any", "fact_ids": ["atlas_steam_pressure"], "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a03", "corpus": "atlas7", "query": "", "expected": ["93.5 degrees Celsius"], "match": "any", "fact_ids": ["atlas_brew_temperature"], "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": "{\"question\":\"What temperature does the PID hold brew water at?\"}"} +{"id": "atlas7_a04", "corpus": "atlas7", "query": "What is the continuous duty rating of the vibratory pump?", "expected": ["52 watts continuous duty"], "match": "any", "fact_ids": ["atlas_pump_rating"], "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a05", "corpus": "atlas7", "query": "What is the hardness threshold?", "expected": ["120 ppm"], "match": "any", "fact_ids": ["atlas_water_hardness"], "dimensions": {"topic": "maintenance", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a06", "corpus": "atlas7", "query": "What part number is the group head gasket?", "expected": ["group head gasket (part MG-311)"], "match": "any", "fact_ids": ["atlas_gasket_part"], "dimensions": {"topic": "maintenance", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a07", "corpus": "atlas7", "query": "What is the duration of the parts warranty?", "expected": ["36-month parts warranty"], "match": "any", "fact_ids": ["atlas_warranty_length"], "dimensions": {"topic": "warranty", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a08", "corpus": "atlas7", "query": "Who manufactures Meridian Coffee Systems?", "expected": ["Meridian Coffee Systems, Tacoma WA"], "match": "any", "fact_ids": ["atlas_manufacturer"], "dimensions": {"topic": "identification", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a09", "corpus": "atlas7", "query": "What is the revision of the Atlas-7 Dual Boiler espresso machine?", "expected": ["Atlas-7 Dual Boiler (2026 revision C)"], "match": "any", "fact_ids": ["atlas_model_revision"], "dimensions": {"topic": "identification", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a10", "corpus": "atlas7", "query": "Where is the serial number located?", "expected": ["engraved under the drip tray on the left rail"], "match": "any", "fact_ids": ["atlas_serial_location"], "dimensions": {"topic": "warranty", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a11", "corpus": "atlas7", "query": "", "expected": ["tolerance of 0.4 degrees"], "match": "any", "fact_ids": ["atlas_temperature_tolerance"], "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": "{\"question\":\"What is the allowable variance in brew temperature for this machine?\"}"} +{"id": "atlas7_a12", "corpus": "atlas7", "query": "Which component should be swapped out to clear the E11 fault code?", "expected": ["Replace sensor part TS-71"], "match": "any", "fact_ids": ["atlas_e11_part"], "dimensions": {"topic": "error_codes", "question_type": "factoid", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a13", "corpus": "atlas7", "query": "How do I fix E42 pump cavitation on the espresso machine?", "expected": ["Prime the pump by running 200 ml", "hot water wand, then power cycle the unit"], "match": "all", "fact_ids": ["atlas_e42_prime", "atlas_e42_procedure"], "dimensions": {"topic": "error_codes", "question_type": "procedural", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a14", "corpus": "atlas7", "query": "How often should I backflush the espresso machine?", "expected": ["Backflushing with Cafiza detergent is recommended weekly"], "match": "any", "fact_ids": ["atlas_backflush"], "dimensions": {"topic": "maintenance", "question_type": "procedural", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a15", "corpus": "atlas7", "query": "How often should I run the descaling cycle if my water contains more than 120 ppm of hardness?", "expected": ["every 60 days when water hardness exceeds", "120 ppm"], "match": "all", "fact_ids": ["atlas_descale_interval", "atlas_water_hardness"], "dimensions": {"topic": "maintenance", "question_type": "procedural", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a16", "corpus": "atlas7", "query": "What steps should I take if my espresso machine stops delivering water?", "expected": ["Clean the inlet mesh filter"], "match": "any", "fact_ids": ["atlas_e57"], "dimensions": {"topic": "error_codes", "question_type": "procedural", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a17", "corpus": "atlas7", "query": "What should an operator do to address steam overpressure?", "expected": ["Check the OPV calibration at 12 bar"], "match": "any", "fact_ids": ["atlas_e23"], "dimensions": {"topic": "error_codes", "question_type": "procedural", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a18", "corpus": "atlas7", "query": "what is the pressure difference between the brew boiler and steam boiler during extraction?", "expected": ["pressure of 9.2 bar", "steam boiler is maintained at 1.45 bar"], "match": "all", "fact_ids": ["atlas_brew_pressure", "atlas_steam_pressure"], "dimensions": {"topic": "specifications", "question_type": "comparative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a19", "corpus": "atlas7", "query": "Between the water hardness threshold that triggers descaling every two months and the time interval for replacing the gasket, which maintenance action requires a longer waiting period?", "expected": ["every 60 days when water hardness exceeds", "replaced every 14 months"], "match": "all", "fact_ids": ["atlas_descale_interval", "atlas_gasket_interval"], "dimensions": {"topic": "maintenance", "question_type": "comparative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a20", "corpus": "atlas7", "query": "", "expected": ["E11: Brew boiler thermistor open circuit", "Check the OPV calibration at 12 bar"], "match": "all", "fact_ids": ["atlas_e11", "atlas_e23"], "dimensions": {"topic": "error_codes", "question_type": "comparative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": "{\"question\":\"What is the relationship between a steam overpressure condition involving an OPV check and brew boiler thermistor issues?\"}"} +{"id": "atlas7_a21", "corpus": "atlas7", "query": "Which third-party descalers void the warranty due to citric acid concentration exceeding 8 percent?", "expected": ["citric acid above 8 percent concentration"], "match": "any", "fact_ids": ["atlas_warranty_void"], "dimensions": {"topic": "warranty", "question_type": "negative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a22", "corpus": "atlas7", "query": "What happens if water hardness is higher than 120 ppm?", "expected": ["120 ppm"], "match": "any", "fact_ids": ["atlas_water_hardness"], "dimensions": {"topic": "maintenance", "question_type": "negative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a23", "corpus": "atlas7", "query": "What is maximum allowed deviation from the target brew temperature for an espresso machine?", "expected": ["tolerance of 0.4 degrees"], "match": "any", "fact_ids": ["atlas_temperature_tolerance"], "dimensions": {"topic": "specifications", "question_type": "negative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "atlas7_a24", "corpus": "atlas7", "query": "under what pressure should the OPV calibration be performed if steam overpressure is detected?", "expected": ["Check the OPV calibration at 12 bar"], "match": "any", "fact_ids": ["atlas_e23"], "dimensions": {"topic": "error_codes", "question_type": "negative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} diff --git a/eval/goldset/_generated/docs.raw.jsonl b/eval/goldset/_generated/docs.raw.jsonl new file mode 100644 index 00000000..7ddab2cf --- /dev/null +++ b/eval/goldset/_generated/docs.raw.jsonl @@ -0,0 +1,24 @@ +{"id": "docs_d01", "corpus": "docs", "query": "What model is used for sentence pruning?", "expected": ["naver/provence-reranker-debertav3-v1"], "match": "any", "fact_ids": ["docs_provence_model"], "dimensions": {"topic": "pruning", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d02", "corpus": "docs", "query": "How many characters does the verifier prompt clamp context to?", "expected": ["clamped to the first 4000 characters"], "match": "any", "fact_ids": ["docs_verifier_context_clamp"], "dimensions": {"topic": "verifier", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d03", "corpus": "docs", "query": "How many overviews does the overview router use?", "expected": ["the first 40 loaded overviews"], "match": "any", "fact_ids": ["docs_triage_overview_cap"], "dimensions": {"topic": "triage", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d04", "corpus": "docs", "query": "How many dimensions does the Qwen3-Embedding-4B produce?", "expected": ["produces 2560-dim vectors"], "match": "any", "fact_ids": ["docs_embedding_dimensions"], "dimensions": {"topic": "embedding_model", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d05", "corpus": "docs", "query": "How many characters is the overview input truncated to?", "expected": ["truncated to 5000 characters"], "match": "any", "fact_ids": ["docs_overview_truncation"], "dimensions": {"topic": "overviews", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d06", "corpus": "docs", "query": "What is the maximum number of sentences in a reply generated by direct_answer?", "expected": ["Caps the reply at 1-2 sentences"], "match": "any", "fact_ids": ["docs_direct_answer_length"], "dimensions": {"topic": "prompts", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d07", "corpus": "docs", "query": "How much does late chunking increase the number of vectors written compared to early chunking?", "expected": ["roughly double the vectors written"], "match": "any", "fact_ids": ["docs_latechunk_cost"], "dimensions": {"topic": "late_chunking", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d08", "corpus": "docs", "query": "What happens to a summary shorter than 5 characters?", "expected": ["a summary shorter than 5 characters is discarded"], "match": "any", "fact_ids": ["docs_enrichment_short_summary"], "dimensions": {"topic": "enrichment", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d09", "corpus": "docs", "query": "How many slots are available in the cache used to store query representations for each search engine?", "expected": ["memoised in a 256-entry"], "match": "any", "fact_ids": ["docs_query_embed_cache"], "dimensions": {"topic": "hybrid_retrieval", "question_type": "factoid", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d10", "corpus": "docs", "query": "How many requests to a language model does each processed unit of information require?", "expected": ["makes two LLM calls per chunk"], "match": "any", "fact_ids": ["docs_graph_two_llm_calls"], "dimensions": {"topic": "graph", "question_type": "factoid", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d11", "corpus": "docs", "query": "What value indicates that a tag could not be appended due to parsing failure?", "expected": ["0 is treated as a parse failure"], "match": "any", "fact_ids": ["docs_verifier_zero_score"], "dimensions": {"topic": "verifier", "question_type": "factoid", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d12", "corpus": "docs", "query": "How do I change the embedding model?", "expected": ["Changing the embedding model requires re-indexing"], "match": "any", "fact_ids": ["docs_reindex_required"], "dimensions": {"topic": "embedding_model", "question_type": "procedural", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d13", "corpus": "docs", "query": "how do I enable pruning in localGPT?", "expected": ["so pruning is off unless a request enables it"], "match": "any", "fact_ids": ["docs_pruning_off_by_default"], "dimensions": {"topic": "pruning", "question_type": "procedural", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d14", "corpus": "docs", "query": "", "expected": ["while the HTTP path always sends"], "match": "any", "fact_ids": ["docs_chunk_size_layering"], "dimensions": {"topic": "chunking", "question_type": "procedural", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": "{\"question\":\"If you are using the HTTP path to generate code instead of the CLI option, how does this alter your output token limit?\"}"} +{"id": "docs_d15", "corpus": "docs", "query": "How can I handle plain text files if they do not go through the standard processing pipeline?", "expected": ["files bypass docling entirely and are wrapped in a fenced code block"], "match": "any", "fact_ids": ["docs_txt_bypasses_docling"], "dimensions": {"topic": "conversion", "question_type": "procedural", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d16", "corpus": "docs", "query": "Is vector search in rag_system a brute-force scan because no ANN index is created?", "expected": ["No ANN index is created", "Vector search is a brute-force scan"], "match": "all", "fact_ids": ["docs_no_ann_index", "docs_brute_force_vector"], "dimensions": {"topic": "vector_index", "question_type": "comparative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d17", "corpus": "docs", "query": "How does switching to a different large language model for both verification and routing stages impact the total inference rounds required per query?", "expected": ["one extra LLM round-trip per answered query", "Both routers run on the utility model"], "match": "all", "fact_ids": ["docs_verifier_cost", "docs_triage_utility_model"], "dimensions": {"topic": "verifier", "question_type": "comparative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d18", "corpus": "docs", "query": "In what fundamental ways do the legacy chunker and its indexing pipeline lack modern architectural features found in distributed systems?", "expected": ["the legacy path has no overlap logic at all", "there is no thread or process pool anywhere in the indexing path"], "match": "all", "fact_ids": ["docs_no_overlap_knob", "docs_indexing_sequential"], "dimensions": {"topic": "chunking", "question_type": "comparative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d19", "corpus": "docs", "query": "Is there a weighted linear blend option available in the localGPT project?", "expected": ["There is no weighted linear blend"], "match": "any", "fact_ids": ["docs_no_weighted_blend"], "dimensions": {"topic": "hybrid_retrieval", "question_type": "negative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d20", "corpus": "docs", "query": "Is there an inline citation marker in localGPT?", "expected": ["There are no inline citation markers"], "match": "any", "fact_ids": ["docs_no_citation_markers"], "dimensions": {"topic": "synthesis", "question_type": "negative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d21", "corpus": "docs", "query": "Is there a regex or keyword stage in the agent router?", "expected": ["There is no regex or keyword stage in the agent"], "match": "any", "fact_ids": ["docs_triage_no_regex"], "dimensions": {"topic": "triage", "question_type": "negative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d22", "corpus": "docs", "query": "can I completely disable the triage mechanism in this system?", "expected": ["There is no global triage on/off switch"], "match": "any", "fact_ids": ["docs_triage_no_switch"], "dimensions": {"topic": "triage", "question_type": "negative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d23", "corpus": "docs", "query": "Are there instances where contexts retrieved as part of neighbor expansion are subsequently removed?", "expected": ["the freshly added neighbours are filtered back out"], "match": "any", "fact_ids": ["docs_expansion_filtered_out"], "dimensions": {"topic": "context_expansion", "question_type": "negative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "docs_d24", "corpus": "docs", "query": "", "expected": ["Silently dropping or recreating an index would corrupt it"], "match": "any", "fact_ids": ["docs_dimension_mismatch_raises"], "dimensions": {"topic": "vector_index", "question_type": "negative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": "{\"question\":\"What prevents an index from being safely recreated when its dimensions do not match?\"}"} diff --git a/eval/goldset/_generated/hr.raw.jsonl b/eval/goldset/_generated/hr.raw.jsonl new file mode 100644 index 00000000..d966c421 --- /dev/null +++ b/eval/goldset/_generated/hr.raw.jsonl @@ -0,0 +1,24 @@ +{"id": "hr_h01", "corpus": "hr", "query": "How many days of paid annual leave do employees below Grade 7 accrue per calendar year?", "expected": ["23 days of paid annual leave"], "match": "any", "fact_ids": ["hr_annual_below_g7"], "dimensions": {"topic": "annual_leave", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h02", "corpus": "hr", "query": "How many days do Grade 7 and above accrue?", "expected": ["Grade 7 and above accrue 28 days"], "match": "any", "fact_ids": ["hr_annual_g7_plus"], "dimensions": {"topic": "annual_leave", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h03", "corpus": "hr", "query": "At what salary percentage is sick leave paid for the first 12 weeks?", "expected": ["100 percent of base salary for the first 12 weeks"], "match": "any", "fact_ids": ["hr_sick_full_pay"], "dimensions": {"topic": "sick_leave", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h04", "corpus": "hr", "query": "How many weeks of parental leave per child are fully paid?", "expected": ["Parental leave is 18 weeks per child, of which 6 weeks are fully paid"], "match": "any", "fact_ids": ["hr_parental_length"], "dimensions": {"topic": "parental_leave", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h05", "corpus": "hr", "query": "How many days of leave are available for an immediate family member?", "expected": ["5 working days for an immediate family member"], "match": "any", "fact_ids": ["hr_bereavement"], "dimensions": {"topic": "bereavement", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h06", "corpus": "hr", "query": "How many working days is jury service paid in full for per year?", "expected": ["paid in full for up to 15 working days"], "match": "any", "fact_ids": ["hr_jury_duty"], "dimensions": {"topic": "jury_duty", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h07", "corpus": "hr", "query": "How many public holidays does Northwind Robotics recognise?", "expected": ["recognises 9 public holidays"], "match": "any", "fact_ids": ["hr_public_holiday_count"], "dimensions": {"topic": "public_holidays", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h08", "corpus": "hr", "query": "What is the effective date of policy PPL-204 revision 4?", "expected": ["Policy PPL-204, revision 4"], "match": "any", "fact_ids": ["hr_policy_id"], "dimensions": {"topic": "policy_metadata", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h09", "corpus": "hr", "query": "Who owns the Department of People Operations at Northwind Robotics?", "expected": ["Department of People Operations, Northwind Robotics, Gothenburg"], "match": "any", "fact_ids": ["hr_policy_owner"], "dimensions": {"topic": "policy_metadata", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h10", "corpus": "hr", "query": "What is the maximum duration of an unpaid sabbatical?", "expected": ["unpaid sabbatical of up to 90 days"], "match": "any", "fact_ids": ["hr_sabbatical_length"], "dimensions": {"topic": "sabbatical", "question_type": "factoid", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h11", "corpus": "hr", "query": "What date marks the end of the expiration period for rollover leave days?", "expected": ["Carried days expire on 31 March"], "match": "any", "fact_ids": ["hr_carryover_expiry"], "dimensions": {"topic": "annual_leave", "question_type": "factoid", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h12", "corpus": "hr", "query": "If a person takes partial leave for one month and continues after that period, what percentage rate applies to the next two months?", "expected": ["at 60 percent for a further 8 weeks"], "match": "any", "fact_ids": ["hr_sick_reduced_pay"], "dimensions": {"topic": "sick_leave", "question_type": "factoid", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h13", "corpus": "hr", "query": "What is the multiplier applied to a standard wage when an employee works through a public holiday and gets paid?", "expected": ["paid at 1.5 times the normal rate"], "match": "any", "fact_ids": ["hr_public_holiday_pay"], "dimensions": {"topic": "public_holidays", "question_type": "factoid", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h14", "corpus": "hr", "query": "How many working days ahead must requests go through the Kestrel HR portal?", "expected": ["Kestrel HR portal at least 10"], "match": "any", "fact_ids": ["hr_request_notice"], "dimensions": {"topic": "requesting_leave", "question_type": "procedural", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h15", "corpus": "hr", "query": "What happens if I need to take more than 4 consecutive working days off due to a medical issue?", "expected": ["exceeds 4 consecutive working days"], "match": "any", "fact_ids": ["hr_medical_certificate"], "dimensions": {"topic": "sick_leave", "question_type": "procedural", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h16", "corpus": "hr", "query": "What conditions must be met and what steps are needed to take a leave away from work?", "expected": ["at least 4 years of continuous service", "require 60 days written", "approved by the Head of People Operations"], "match": "all", "fact_ids": ["hr_sabbatical_eligibility", "hr_sabbatical_notice", "hr_sabbatical_approver"], "dimensions": {"topic": "sabbatical", "question_type": "procedural", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h17", "corpus": "hr", "query": "How can an employee obtain authorization for a run of more than ten back-to-back workdays?", "expected": ["consecutive working days additionally requires written approval"], "match": "any", "fact_ids": ["hr_director_approval"], "dimensions": {"topic": "requesting_leave", "question_type": "procedural", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h18", "corpus": "hr", "query": "", "expected": ["23 days of paid annual leave", "Grade 7 and above accrue 28 days"], "match": "all", "fact_ids": ["hr_annual_below_g7", "hr_annual_g7_plus"], "dimensions": {"topic": "annual_leave", "question_type": "comparative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": "{\"question\":\"What is the difference in paid annual leave accrual between an employee below Grade 7 and one at that grade?\"}"} +{"id": "hr_h19", "corpus": "hr", "query": "How does the paid sick leave percentage for employees in their second month of absence compare to that after 20 weeks?", "expected": ["100 percent of base salary for the first 12 weeks", "at 60 percent for a further 8 weeks"], "match": "all", "fact_ids": ["hr_sick_full_pay", "hr_sick_reduced_pay"], "dimensions": {"topic": "sick_leave", "question_type": "comparative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h20", "corpus": "hr", "query": "How do unused annual leave days behave regarding both the maximum amount that can be rolled over and their expiration date?", "expected": ["maximum of 5 unused annual leave days may be carried", "Carried days expire on 31 March"], "match": "all", "fact_ids": ["hr_carryover_cap", "hr_carryover_expiry"], "dimensions": {"topic": "annual_leave", "question_type": "comparative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h21", "corpus": "hr", "query": "Are agency contractors covered by the leave and absence policy?", "expected": ["Contractors engaged through an agency"], "match": "any", "fact_ids": ["hr_contractors_excluded"], "dimensions": {"topic": "exclusions", "question_type": "negative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h22", "corpus": "hr", "query": "What happens to my annual leave if I resign before completing six months of service?", "expected": ["not paid out on resignation"], "match": "any", "fact_ids": ["hr_resignation_payout"], "dimensions": {"topic": "exclusions", "question_type": "negative", "difficulty": "easy"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h23", "corpus": "hr", "query": "Can a parent utilize their leave entitlement after the child reaches three years of age?", "expected": ["before the child's third birthday"], "match": "any", "fact_ids": ["hr_parental_deadline"], "dimensions": {"topic": "parental_leave", "question_type": "negative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} +{"id": "hr_h24", "corpus": "hr", "query": "Can a user schedule more than three distinct parental leave segments without violating the rules?", "expected": ["no more than 3 separate blocks"], "match": "any", "fact_ids": ["hr_parental_blocks"], "dimensions": {"topic": "parental_leave", "question_type": "negative", "difficulty": "hard"}, "generator_model": "qwen3.5:4b", "raw_response": null} diff --git a/eval/goldset/acquisition.jsonl b/eval/goldset/acquisition.jsonl new file mode 100644 index 00000000..906ab529 --- /dev/null +++ b/eval/goldset/acquisition.jsonl @@ -0,0 +1,24 @@ +{"id": "acq_q01", "corpus": "acq", "query": "What is the total purchase price for the StartupXYZ acquisition?", "expected": ["means $45,000,000 USD as detailed in Exhibit A"], "match": "any", "fact_ids": ["acq_purchase_price"], "answer": "The purchase price is $45,000,000 USD, detailed in Exhibit A - Financial Terms.", "expected_sources": ["01_acquisition_agreement.pdf"], "anchor_doc": "01_acquisition_agreement.pdf", "multi_document": false, "dimensions": {"topic": "purchase_price", "question_type": "factoid", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "TEST_QUESTIONS.md Q1, verbatim", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "means $45,000,000 USD as detailed in Exhibit A", "found_in": ["01_acquisition_agreement.pdf"]}]}}} +{"id": "acq_q02", "corpus": "acq", "query": "When was the mutual non-disclosure agreement between TechCorp and StartupXYZ signed?", "expected": ["entered into as of October 1, 2024"], "match": "any", "fact_ids": ["nda_execution_date"], "answer": "The mutual NDA was entered into as of October 1, 2024.", "expected_sources": ["07_nda.pdf"], "anchor_doc": "07_nda.pdf", "multi_document": false, "dimensions": {"topic": "confidentiality", "question_type": "factoid", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "TEST_QUESTIONS.md Q2, verbatim", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "entered into as of October 1, 2024", "found_in": ["07_nda.pdf"]}]}}} +{"id": "acq_q03", "corpus": "acq", "query": "How many United States patents does StartupXYZ own?", "expected": ["StartupXYZ owns 12 U.S. patents as listed in Schedule 1 - IP Assets"], "match": "any", "fact_ids": ["ip_patent_count"], "answer": "StartupXYZ owns 12 U.S. patents, listed in Schedule 1 - IP Assets.", "expected_sources": ["03_ip_certification.pdf"], "anchor_doc": "03_ip_certification.pdf", "multi_document": false, "dimensions": {"topic": "intellectual_property", "question_type": "factoid", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "TEST_QUESTIONS.md Q3, adapted", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "StartupXYZ owns 12 U.S. patents as listed in Schedule 1 - IP Assets", "found_in": ["03_ip_certification.pdf"]}]}}} +{"id": "acq_q04", "corpus": "acq", "query": "What proportion of the target company's turnover comes from its single biggest client?", "expected": ["Largest customer (MegaCorp) accounts for 28% of revenue"], "match": "any", "fact_ids": ["dd_megacorp_concentration"], "answer": "The largest customer, MegaCorp, accounts for 28% of revenue.", "expected_sources": ["02_due_diligence_report.pdf"], "anchor_doc": null, "multi_document": false, "dimensions": {"topic": "customer_concentration", "question_type": "factoid", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Largest customer (MegaCorp) accounts for 28% of revenue", "found_in": ["02_due_diligence_report.pdf"]}]}}} +{"id": "acq_q05", "corpus": "acq", "query": "Which categories of information are carved out of the confidentiality obligations in the NDA?", "expected": ["Is or becomes publicly available through no fault of the receiving Party", "Is independently developed without use of Confidential Information"], "match": "all", "fact_ids": ["nda_exclusion_public", "nda_exclusion_independent"], "answer": "Information that is publicly available through no fault of the receiver, or independently developed without using confidential information (among four exclusions).", "expected_sources": ["07_nda.pdf", "07_nda.pdf"], "anchor_doc": "07_nda.pdf", "multi_document": false, "dimensions": {"topic": "confidentiality", "question_type": "negative", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Is or becomes publicly available through no fault of the receiving Party", "found_in": ["07_nda.pdf"]}, {"expected": "Is independently developed without use of Confidential Information", "found_in": ["07_nda.pdf"]}]}}} +{"id": "acq_q06", "corpus": "acq", "query": "What must each side do with the other's confidential material once it is asked for or the arrangement ends?", "expected": ["each Party shall return or destroy all Confidential Information"], "match": "any", "fact_ids": ["nda_return_of_materials"], "answer": "Each party must return or destroy all confidential information, except what it must keep for legal or regulatory purposes.", "expected_sources": ["07_nda.pdf"], "anchor_doc": "07_nda.pdf", "multi_document": false, "dimensions": {"topic": "confidentiality", "question_type": "procedural", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "each Party shall return or destroy all Confidential Information", "found_in": ["07_nda.pdf"]}]}}} +{"id": "acq_q07", "corpus": "acq", "query": "How much did the parties pay to file under Hart-Scott-Rodino?", "expected": ["HSR Filing Fee: $30,000"], "match": "any", "fact_ids": ["reg_filing_fee"], "answer": "The HSR filing fee was $30,000.", "expected_sources": ["08_regulatory_approval.pdf"], "anchor_doc": "08_regulatory_approval.pdf", "multi_document": false, "dimensions": {"topic": "regulatory", "question_type": "factoid", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "HSR Filing Fee: $30,000", "found_in": ["08_regulatory_approval.pdf"]}]}}} +{"id": "acq_q08", "corpus": "acq", "query": "Which subject did the seller's counsel refuse to give a view on in its opinion letter?", "expected": ["We express no opinion on tax matters"], "match": "any", "fact_ids": ["legal_tax_carveout"], "answer": "Tax matters โ€” the opinion expressly excludes them and points to a separate tax opinion.", "expected_sources": ["06_legal_opinion.pdf"], "anchor_doc": "06_legal_opinion.pdf", "multi_document": false, "dimensions": {"topic": "legal_opinion", "question_type": "negative", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "We express no opinion on tax matters", "found_in": ["06_legal_opinion.pdf"]}]}}} +{"id": "acq_q09", "corpus": "acq", "query": "How does the borrowing the reviewers first disclosed compare with the extra borrowing found later, and what was the later amount?", "expected": ["Outstanding debt: $1.5 million", "Additional identified debt: $175,000 (capital lease obligations)"], "match": "all", "fact_ids": ["dd_disclosed_debt", "fin_extra_debt"], "answer": "Due diligence disclosed $1.5 million of debt; a further $175,000 of capital lease obligations was identified afterwards.", "expected_sources": ["02_due_diligence_report.pdf", "05_financial_adjustments.pdf"], "anchor_doc": null, "multi_document": true, "dimensions": {"topic": "financials", "question_type": "comparative", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored (multi-document)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Outstanding debt: $1.5 million", "found_in": ["02_due_diligence_report.pdf"]}, {"expected": "Additional identified debt: $175,000 (capital lease obligations)", "found_in": ["05_financial_adjustments.pdf"]}]}}} +{"id": "acq_q10", "corpus": "acq", "query": "How much annual revenue was flagged as being at risk from the largest customer, and did that customer end up agreeing to the change of control?", "expected": ["Impact if materialized: $3.4M annual revenue at risk", "MegaCorp Inc. - OBTAINED"], "match": "all", "fact_ids": ["risk_megacorp_impact", "cons_megacorp_status"], "answer": "$3.4M of annual revenue was at risk; MegaCorp's consent was obtained on February 10, 2025.", "expected_sources": ["04_risk_assessment.pdf", "09_customer_consents.pdf"], "anchor_doc": null, "multi_document": true, "dimensions": {"topic": "customer_concentration", "question_type": "comparative", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "TEST_QUESTIONS.md Q6/Q13, adapted (multi-document)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Impact if materialized: $3.4M annual revenue at risk", "found_in": ["04_risk_assessment.pdf"]}, {"expected": "MegaCorp Inc. - OBTAINED", "found_in": ["09_customer_consents.pdf"]}]}}} +{"id": "acq_q11", "corpus": "acq", "query": "The finance team revised the cash portion of the deal. What figure did they land on, and does the closing paperwork carry the same number?", "expected": ["Cash at closing: $28,330,000 (adjusted)", "Cash payment: $28,330,000"], "match": "all", "fact_ids": ["fin_revised_cash", "close_cash_payment"], "answer": "The revised cash at closing is $28,330,000, and the closing checklist carries the same figure.", "expected_sources": ["05_financial_adjustments.pdf", "10_closing_checklist.pdf"], "anchor_doc": null, "multi_document": true, "dimensions": {"topic": "purchase_price", "question_type": "comparative", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored (multi-document)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Cash at closing: $28,330,000 (adjusted)", "found_in": ["05_financial_adjustments.pdf"]}, {"expected": "Cash payment: $28,330,000", "found_in": ["10_closing_checklist.pdf"]}]}}} +{"id": "acq_q12", "corpus": "acq", "query": "How large is the target's workforce and how is it split between functions?", "expected": ["Total employees: 47 (32 engineering, 8 sales, 7 operations)"], "match": "any", "fact_ids": ["dd_headcount"], "answer": "47 employees: 32 in engineering, 8 in sales, 7 in operations.", "expected_sources": ["02_due_diligence_report.pdf"], "anchor_doc": "02_due_diligence_report.pdf", "multi_document": false, "dimensions": {"topic": "employees", "question_type": "factoid", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Total employees: 47 (32 engineering, 8 sales, 7 operations)", "found_in": ["02_due_diligence_report.pdf"]}]}}} +{"id": "acq_q13", "corpus": "acq", "query": "Article IV of the Acquisition Agreement makes the buyer's obligation to close conditional on a regulatory approval. On what date was that condition satisfied?", "expected": ["Early Termination Granted: January 28, 2025"], "match": "any", "fact_ids": ["reg_termination_date"], "answer": "January 28, 2025, when the FTC granted early termination of the HSR waiting period.", "expected_sources": ["08_regulatory_approval.pdf"], "anchor_doc": "01_acquisition_agreement.pdf", "multi_document": false, "dimensions": {"topic": "regulatory", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored (cross-reference: 01 Article IV 4.1(a) -> 08)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Early Termination Granted: January 28, 2025", "found_in": ["08_regulatory_approval.pdf"]}]}}} +{"id": "acq_q14", "corpus": "acq", "query": "Employee matters under the Acquisition Agreement are pushed to Schedule 3. What was the estimated price tag of the retention packages contemplated there?", "expected": ["Estimated cost: $2.5M in retention bonuses"], "match": "any", "fact_ids": ["risk_retention_cost"], "answer": "About $2.5M in retention bonuses.", "expected_sources": ["04_risk_assessment.pdf"], "anchor_doc": "01_acquisition_agreement.pdf", "multi_document": false, "dimensions": {"topic": "employees", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored (cross-reference: 01 Schedule 3 -> 04)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Estimated cost: $2.5M in retention bonuses", "found_in": ["04_risk_assessment.pdf"]}]}}} +{"id": "acq_q15", "corpus": "acq", "query": "The Acquisition Agreement fixes the price by reference to Exhibit A - Financial Terms. After the recommended write-downs, what did the price become?", "expected": ["Adjusted Purchase Price: $43,330,000"], "match": "any", "fact_ids": ["fin_adjusted_price"], "answer": "$43,330,000 after working capital, debt and revenue recognition adjustments.", "expected_sources": ["05_financial_adjustments.pdf"], "anchor_doc": "01_acquisition_agreement.pdf", "multi_document": false, "dimensions": {"topic": "purchase_price", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "TEST_QUESTIONS.md Q5/Q11, adapted (cross-reference: 01 Exhibit A -> 05)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Adjusted Purchase Price: $43,330,000", "found_in": ["05_financial_adjustments.pdf"]}]}}} +{"id": "acq_q16", "corpus": "acq", "query": "In the Acquisition Agreement the seller warrants that its intellectual property is unencumbered, as confirmed by an outside certification. What did that certification conclude about liens on the patents?", "expected": ["All patents are valid, enforceable, and free of liens or encumbrances"], "match": "any", "fact_ids": ["ip_encumbrances"], "answer": "That all the patents are valid, enforceable and free of liens or encumbrances.", "expected_sources": ["03_ip_certification.pdf"], "anchor_doc": "01_acquisition_agreement.pdf", "multi_document": false, "dimensions": {"topic": "intellectual_property", "question_type": "factoid", "difficulty": "easy", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored (cross-reference: 01 s3.2 -> 03)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "All patents are valid, enforceable, and free of liens or encumbrances", "found_in": ["03_ip_certification.pdf"]}]}}} +{"id": "acq_q17", "corpus": "acq", "query": "The NDA treats the target's financial information as confidential and points to where that information sits. What top-line revenue did that material report for FY2024?", "expected": ["Revenue for FY2024: $12.3 million (growth of 45% YoY)"], "match": "any", "fact_ids": ["dd_revenue"], "answer": "$12.3 million, 45% growth year on year.", "expected_sources": ["02_due_diligence_report.pdf"], "anchor_doc": "07_nda.pdf", "multi_document": false, "dimensions": {"topic": "financials", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored (cross-reference: 07 s1 -> 02)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Revenue for FY2024: $12.3 million (growth of 45% YoY)", "found_in": ["02_due_diligence_report.pdf"]}]}}} +{"id": "acq_q18", "corpus": "acq", "query": "The risk memo rates a single pending patent application as a low-priority item. What is the serial number of that application?", "expected": ["one pending patent application (Application No. 17/456,789)"], "match": "any", "fact_ids": ["ip_pending_application"], "answer": "Application No. 17/456,789, for an advanced federated learning system.", "expected_sources": ["03_ip_certification.pdf"], "anchor_doc": "04_risk_assessment.pdf", "multi_document": false, "dimensions": {"topic": "intellectual_property", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "TEST_QUESTIONS.md Q7, adapted (cross-reference: 04 s3.1 -> 03)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "one pending patent application (Application No. 17/456,789)", "found_in": ["03_ip_certification.pdf"]}]}}} +{"id": "acq_q19", "corpus": "acq", "query": "The legal opinion excepts change-of-control provisions from its no-conflicts view. For each affected customer, what is the current consent status?", "expected": ["MegaCorp Inc. - OBTAINED", "DataFlow Systems - OBTAINED", "CloudTech Partners - PENDING"], "match": "all", "fact_ids": ["cons_megacorp_status", "cons_dataflow_status", "cons_cloudtech_status"], "answer": "MegaCorp obtained, DataFlow Systems obtained, CloudTech Partners still pending.", "expected_sources": ["09_customer_consents.pdf", "09_customer_consents.pdf", "09_customer_consents.pdf"], "anchor_doc": "06_legal_opinion.pdf", "multi_document": false, "dimensions": {"topic": "consents", "question_type": "comparative", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "TEST_QUESTIONS.md Q6/Q12, adapted (cross-reference: 06 s3 -> 09)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "MegaCorp Inc. - OBTAINED", "found_in": ["09_customer_consents.pdf"]}, {"expected": "DataFlow Systems - OBTAINED", "found_in": ["09_customer_consents.pdf"]}, {"expected": "CloudTech Partners - PENDING", "found_in": ["09_customer_consents.pdf"]}]}}} +{"id": "acq_q20", "corpus": "acq", "query": "The due diligence report says a Hart-Scott-Rodino filing is needed and refers the timeline elsewhere. On what date was the filing submitted?", "expected": ["Filing Date: January 10, 2025"], "match": "any", "fact_ids": ["reg_filing_date"], "answer": "January 10, 2025.", "expected_sources": ["08_regulatory_approval.pdf"], "anchor_doc": "02_due_diligence_report.pdf", "multi_document": false, "dimensions": {"topic": "regulatory", "question_type": "factoid", "difficulty": "easy", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored (cross-reference: 02 s5.2 -> 08)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Filing Date: January 10, 2025", "found_in": ["08_regulatory_approval.pdf"]}]}}} +{"id": "acq_q21", "corpus": "acq", "query": "The closing checklist calls for an escrow agreement tied to Exhibit C. Over what period is that escrow released?", "expected": ["Escrow: $1,300,000 (18-month release schedule)"], "match": "any", "fact_ids": ["fin_escrow_release"], "answer": "The $1,300,000 escrow is released over 18 months.", "expected_sources": ["05_financial_adjustments.pdf"], "anchor_doc": "10_closing_checklist.pdf", "multi_document": false, "dimensions": {"topic": "escrow", "question_type": "procedural", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored (cross-reference: 10 s II.C -> 05)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Escrow: $1,300,000 (18-month release schedule)", "found_in": ["05_financial_adjustments.pdf"]}]}}} +{"id": "acq_q22", "corpus": "acq", "query": "The FTC's early-termination letter tells the parties they may proceed to closing. Where is that closing scheduled to take place?", "expected": ["Closing Location: Wilson & Partners LLP, San Francisco"], "match": "any", "fact_ids": ["close_location"], "answer": "At Wilson & Partners LLP in San Francisco.", "expected_sources": ["10_closing_checklist.pdf"], "anchor_doc": "08_regulatory_approval.pdf", "multi_document": false, "dimensions": {"topic": "timeline", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "TEST_QUESTIONS.md Q9, adapted (cross-reference: 08 -> 10)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Closing Location: Wilson & Partners LLP, San Francisco", "found_in": ["10_closing_checklist.pdf"]}]}}} +{"id": "acq_q23", "corpus": "acq", "query": "The Acquisition Agreement sets out a cash payment and makes closing conditional on antitrust clearance. What is the revised cash figure, and when did the clearance come through?", "expected": ["Cash at closing: $28,330,000 (adjusted)", "Early Termination Granted: January 28, 2025"], "match": "all", "fact_ids": ["fin_revised_cash", "reg_termination_date"], "answer": "Cash at closing was revised to $28,330,000, and early termination was granted on January 28, 2025.", "expected_sources": ["05_financial_adjustments.pdf", "08_regulatory_approval.pdf"], "anchor_doc": "01_acquisition_agreement.pdf", "multi_document": true, "dimensions": {"topic": "purchase_price", "question_type": "comparative", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "TEST_QUESTIONS.md Q10, adapted (cross-reference + multi-document: 01 -> 05, 08)", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "Cash at closing: $28,330,000 (adjusted)", "found_in": ["05_financial_adjustments.pdf"]}, {"expected": "Early Termination Granted: January 28, 2025", "found_in": ["08_regulatory_approval.pdf"]}]}}} +{"id": "acq_q24", "corpus": "acq", "query": "What does early termination of the waiting period not prevent the Commission from doing?", "expected": ["does not preclude the Commission from taking any action"], "match": "any", "fact_ids": ["reg_no_preclusion"], "answer": "It does not preclude the Commission from taking any action it deems necessary to protect competition.", "expected_sources": ["08_regulatory_approval.pdf"], "anchor_doc": "08_regulatory_approval.pdf", "multi_document": false, "dimensions": {"topic": "regulatory", "question_type": "negative", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-09", "provenance": "hand-authored", "checks": {"expected_in_source": true, "no_verbatim_leak": true, "expected_documents": [{"expected": "does not preclude the Commission from taking any action", "found_in": ["08_regulatory_approval.pdf"]}]}}} diff --git a/eval/goldset/atlas7.jsonl b/eval/goldset/atlas7.jsonl new file mode 100644 index 00000000..87dfd5ff --- /dev/null +++ b/eval/goldset/atlas7.jsonl @@ -0,0 +1,24 @@ +{"id": "atlas7_a01", "corpus": "atlas7", "query": "What is the pressure of the brew boiler during extraction?", "expected": ["pressure of 9.2 bar"], "match": "any", "fact_ids": ["atlas_brew_pressure"], "answer": "The brew boiler runs at 9.2 bar during extraction.", "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What is the pressure of the brew boiler during extraction?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a02", "corpus": "atlas7", "query": "What pressure is the steam boiler held at?", "expected": ["steam boiler is maintained at 1.45 bar"], "match": "any", "fact_ids": ["atlas_steam_pressure"], "answer": "The steam boiler is held at 1.45 bar.", "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "model wrapped its output in literal quote characters", "generated_query": "\"What is the pressure at which the steam boiler is maintained?\"", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a03", "corpus": "atlas7", "query": "What temperature does the PID hold the brew water at?", "expected": ["93.5 degrees Celsius"], "match": "any", "fact_ids": ["atlas_brew_temperature"], "answer": "The PID holds brew water at 93.5 C.", "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "rescued", "reason": "model returned {\"question\": ...}; question taken from that payload", "generated_query": "{\"question\":\"What temperature does the PID hold brew water at?\"}", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a04", "corpus": "atlas7", "query": "What is the continuous duty rating of the vibratory pump?", "expected": ["52 watts continuous duty"], "match": "any", "fact_ids": ["atlas_pump_rating"], "answer": "The vibratory pump is rated 52 W continuous.", "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What is the continuous duty rating of the vibratory pump?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a05", "corpus": "atlas7", "query": "What is the water hardness threshold for this machine?", "expected": ["120 ppm"], "match": "any", "fact_ids": ["atlas_water_hardness"], "answer": "The hardness threshold is 120 ppm.", "dimensions": {"topic": "maintenance", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "'What is the hardness threshold?' had no anchor to the document at all", "generated_query": "What is the hardness threshold?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a06", "corpus": "atlas7", "query": "What part number is the group head gasket?", "expected": ["group head gasket (part MG-311)"], "match": "any", "fact_ids": ["atlas_gasket_part"], "answer": "The group head gasket is part MG-311.", "dimensions": {"topic": "maintenance", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What part number is the group head gasket?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a07", "corpus": "atlas7", "query": "What is the duration of the parts warranty?", "expected": ["36-month parts warranty"], "match": "any", "fact_ids": ["atlas_warranty_length"], "answer": "The parts warranty is 36 months.", "dimensions": {"topic": "warranty", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What is the duration of the parts warranty?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a08", "corpus": "atlas7", "query": "Who makes the Atlas-7, and where are they based?", "expected": ["Meridian Coffee Systems, Tacoma WA"], "match": "any", "fact_ids": ["atlas_manufacturer"], "answer": "Made by Meridian Coffee Systems of Tacoma, WA.", "dimensions": {"topic": "identification", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question asked who manufactures the manufacturer", "generated_query": "Who manufactures Meridian Coffee Systems?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a09", "corpus": "atlas7", "query": "What is the revision of the Atlas-7 Dual Boiler espresso machine?", "expected": ["Atlas-7 Dual Boiler (2026 revision C)"], "match": "any", "fact_ids": ["atlas_model_revision"], "answer": "The model is the Atlas-7 Dual Boiler, 2026 revision C.", "dimensions": {"topic": "identification", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What is the revision of the Atlas-7 Dual Boiler espresso machine?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a10", "corpus": "atlas7", "query": "Where is the serial number located?", "expected": ["engraved under the drip tray on the left rail"], "match": "any", "fact_ids": ["atlas_serial_location"], "answer": "The serial number is under the drip tray on the left rail.", "dimensions": {"topic": "warranty", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "Where is the serial number located?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a11", "corpus": "atlas7", "query": "How tightly is the brew water temperature controlled?", "expected": ["tolerance of 0.4 degrees"], "match": "any", "fact_ids": ["atlas_temperature_tolerance"], "answer": "Brew temperature tolerance is 0.4 degrees.", "dimensions": {"topic": "specifications", "question_type": "factoid", "difficulty": "hard"}, "verification": {"verdict": "rescued", "reason": "model returned {\"question\": ...}", "generated_query": "{\"question\":\"What is the allowable variance in brew temperature for this machine?\"}", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a12", "corpus": "atlas7", "query": "Which component should be swapped out to clear the E11 fault code?", "expected": ["Replace sensor part TS-71"], "match": "any", "fact_ids": ["atlas_e11_part"], "answer": "E11 is fixed by replacing sensor TS-71.", "dimensions": {"topic": "error_codes", "question_type": "factoid", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "Which component should be swapped out to clear the E11 fault code?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a13", "corpus": "atlas7", "query": "How do I fix E42 pump cavitation on the espresso machine?", "expected": ["Prime the pump by running 200 ml", "hot water wand, then power cycle the unit"], "match": "all", "fact_ids": ["atlas_e42_prime", "atlas_e42_procedure"], "answer": "E42 is pump cavitation; prime with 200 ml. Prime via the hot water wand, then power cycle.", "dimensions": {"topic": "error_codes", "question_type": "procedural", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How do I fix E42 pump cavitation on the espresso machine?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a14", "corpus": "atlas7", "query": "How often should I backflush the espresso machine?", "expected": ["Backflushing with Cafiza detergent is recommended weekly"], "match": "any", "fact_ids": ["atlas_backflush"], "answer": "Weekly backflush with Cafiza.", "dimensions": {"topic": "maintenance", "question_type": "procedural", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How often should I backflush the espresso machine?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a15", "corpus": "atlas7", "query": "How often should I run the descaling cycle if my water contains more than 120 ppm of hardness?", "expected": ["every 60 days when water hardness exceeds", "120 ppm"], "match": "all", "fact_ids": ["atlas_descale_interval", "atlas_water_hardness"], "answer": "Descale every 60 days above the hardness threshold. The hardness threshold is 120 ppm.", "dimensions": {"topic": "maintenance", "question_type": "procedural", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How often should I run the descaling cycle if my water contains more than 120 ppm of hardness?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a16", "corpus": "atlas7", "query": "The machine is not registering any water flow at all - what should I check?", "expected": ["Clean the inlet mesh filter"], "match": "any", "fact_ids": ["atlas_e57"], "answer": "E57 is zero flow-meter pulses; clean the inlet mesh filter.", "dimensions": {"topic": "error_codes", "question_type": "procedural", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "'stops delivering water' was ambiguous between the E42 and E57 procedures", "generated_query": "What steps should I take if my espresso machine stops delivering water?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a17", "corpus": "atlas7", "query": "What should an operator do to address steam overpressure?", "expected": ["Check the OPV calibration at 12 bar"], "match": "any", "fact_ids": ["atlas_e23"], "answer": "E23 is steam overpressure; check OPV calibration at 12 bar.", "dimensions": {"topic": "error_codes", "question_type": "procedural", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What should an operator do to address steam overpressure?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a18", "corpus": "atlas7", "query": "what is the pressure difference between the brew boiler and steam boiler during extraction?", "expected": ["pressure of 9.2 bar", "steam boiler is maintained at 1.45 bar"], "match": "all", "fact_ids": ["atlas_brew_pressure", "atlas_steam_pressure"], "answer": "The brew boiler runs at 9.2 bar during extraction. The steam boiler is held at 1.45 bar.", "dimensions": {"topic": "specifications", "question_type": "comparative", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "what is the pressure difference between the brew boiler and steam boiler during extraction?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a19", "corpus": "atlas7", "query": "Which needs doing more often on this machine: descaling, or replacing the group head gasket?", "expected": ["every 60 days when water hardness exceeds", "replaced every 14 months"], "match": "all", "fact_ids": ["atlas_descale_interval", "atlas_gasket_interval"], "answer": "Descale every 60 days above the hardness threshold. The gasket is replaced every 14 months.", "dimensions": {"topic": "maintenance", "question_type": "comparative", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "generated comparative leaked the descaling interval into the question", "generated_query": "Between the water hardness threshold that triggers descaling every two months and the time interval for replacing the gasket, which maintenance action requires a longer waiting period?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a20", "corpus": "atlas7", "query": "Which error code points to a failed temperature sensor, and which one points to excess steam pressure?", "expected": ["E11: Brew boiler thermistor open circuit", "Check the OPV calibration at 12 bar"], "match": "all", "fact_ids": ["atlas_e11", "atlas_e23"], "answer": "E11 is a brew boiler thermistor open circuit. E23 is steam overpressure; check OPV calibration at 12 bar.", "dimensions": {"topic": "error_codes", "question_type": "comparative", "difficulty": "hard"}, "verification": {"verdict": "rescued", "reason": "model returned a truncated {\"question\": ...} payload", "generated_query": "{\"question\":\"What is the relationship between a steam overpressure condition involving an OPV check and brew boiler thermistor issues?\"}", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a21", "corpus": "atlas7", "query": "What kind of descaling product will void the warranty?", "expected": ["citric acid above 8 percent concentration"], "match": "any", "fact_ids": ["atlas_warranty_void"], "answer": "Third-party descalers over 8 percent citric acid void the warranty.", "dimensions": {"topic": "warranty", "question_type": "negative", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question contained the expected answer text verbatim", "generated_query": "Which third-party descalers void the warranty due to citric acid concentration exceeding 8 percent?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a22", "corpus": "atlas7", "query": "Will a warranty claim be accepted without a serial number, and where do I find it?", "expected": ["engraved under the drip tray on the left rail"], "match": "any", "fact_ids": ["atlas_serial_location"], "answer": "The serial number is under the drip tray on the left rail.", "dimensions": {"topic": "warranty", "question_type": "negative", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question restated '120 ppm'; re-anchored to the serial-number requirement", "generated_query": "What happens if water hardness is higher than 120 ppm?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a23", "corpus": "atlas7", "query": "Is there a limit on how far the brew temperature may drift before it is out of spec?", "expected": ["tolerance of 0.4 degrees"], "match": "any", "fact_ids": ["atlas_temperature_tolerance"], "answer": "Brew temperature tolerance is 0.4 degrees.", "dimensions": {"topic": "specifications", "question_type": "negative", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "reworded so it is not a near-duplicate of a11's phrasing", "generated_query": "What is maximum allowed deviation from the target brew temperature for an espresso machine?", "generator_model": "qwen3.5:4b"}} +{"id": "atlas7_a24", "corpus": "atlas7", "query": "under what pressure should the OPV calibration be performed if steam overpressure is detected?", "expected": ["Check the OPV calibration at 12 bar"], "match": "any", "fact_ids": ["atlas_e23"], "answer": "E23 is steam overpressure; check OPV calibration at 12 bar.", "dimensions": {"topic": "error_codes", "question_type": "negative", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "under what pressure should the OPV calibration be performed if steam overpressure is detected?", "generator_model": "qwen3.5:4b"}} diff --git a/eval/goldset/docs.jsonl b/eval/goldset/docs.jsonl new file mode 100644 index 00000000..cd8f5ee5 --- /dev/null +++ b/eval/goldset/docs.jsonl @@ -0,0 +1,24 @@ +{"id": "docs_d01", "corpus": "docs", "query": "What model is used for sentence pruning?", "expected": ["naver/provence-reranker-debertav3-v1"], "match": "any", "fact_ids": ["docs_provence_model"], "answer": "Sentence pruning uses naver/provence-reranker-debertav3-v1.", "dimensions": {"topic": "pruning", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What model is used for sentence pruning?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d02", "corpus": "docs", "query": "How many characters does the verifier prompt clamp context to?", "expected": ["clamped to the first 4000 characters"], "match": "any", "fact_ids": ["docs_verifier_context_clamp"], "answer": "The verifier prompt clamps context to the first 4000 characters.", "dimensions": {"topic": "verifier", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many characters does the verifier prompt clamp context to?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d03", "corpus": "docs", "query": "How many overviews does the overview router use?", "expected": ["the first 40 loaded overviews"], "match": "any", "fact_ids": ["docs_triage_overview_cap"], "answer": "The overview router uses the first 40 loaded overviews.", "dimensions": {"topic": "triage", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many overviews does the overview router use?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d04", "corpus": "docs", "query": "How many dimensions does the Qwen3-Embedding-4B produce?", "expected": ["produces 2560-dim vectors"], "match": "any", "fact_ids": ["docs_embedding_dimensions"], "answer": "Qwen3-Embedding-4B is 2560-dim, the 0.6B is 1024-dim; not interchangeable.", "dimensions": {"topic": "embedding_model", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many dimensions does the Qwen3-Embedding-4B produce?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d05", "corpus": "docs", "query": "How many characters is the overview input truncated to?", "expected": ["truncated to 5000 characters"], "match": "any", "fact_ids": ["docs_overview_truncation"], "answer": "Overview input is the first N chunks truncated to 5000 characters.", "dimensions": {"topic": "overviews", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many characters is the overview input truncated to?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d06", "corpus": "docs", "query": "What is the maximum number of sentences in a reply generated by direct_answer?", "expected": ["Caps the reply at 1-2 sentences"], "match": "any", "fact_ids": ["docs_direct_answer_length"], "answer": "The direct_answer prompt caps the reply at 1-2 sentences.", "dimensions": {"topic": "prompts", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What is the maximum number of sentences in a reply generated by direct_answer?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d07", "corpus": "docs", "query": "How many extra vectors does turning on late chunking write?", "expected": ["roughly double the vectors written"], "match": "any", "fact_ids": ["docs_latechunk_cost"], "answer": "Late chunking costs a second copy of the embedder and roughly double the vectors.", "dimensions": {"topic": "late_chunking", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question compared against 'early chunking', a term the docs never use", "generated_query": "How much does late chunking increase the number of vectors written compared to early chunking?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d08", "corpus": "docs", "query": "What does the enricher do when the model returns an almost-empty summary?", "expected": ["a summary shorter than 5 characters is discarded"], "match": "any", "fact_ids": ["docs_enrichment_short_summary"], "answer": "An enrichment summary under 5 characters is discarded and the chunk is indexed unenriched.", "dimensions": {"topic": "enrichment", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question restated 'shorter than 5 characters'", "generated_query": "What happens to a summary shorter than 5 characters?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d09", "corpus": "docs", "query": "How large is the per-retriever cache that stores previously embedded queries?", "expected": ["memoised in a 256-entry"], "match": "any", "fact_ids": ["docs_query_embed_cache"], "answer": "The query embedding is memoised in a 256-entry lru_cache per retriever.", "dimensions": {"topic": "hybrid_retrieval", "question_type": "factoid", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "generated question invented 'for each search engine'", "generated_query": "How many slots are available in the cache used to store query representations for each search engine?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d10", "corpus": "docs", "query": "What stops two embedding models with the same vector width from silently corrupting an index?", "expected": ["records the embedding model that wrote it", "EmbedderMismatchError"], "match": "any", "fact_ids": ["docs_identity_marker"], "answer": "Each table records the embedding model that wrote it (and a normalized flag); a mismatch raises EmbedderMismatchError at index and query time.", "dimensions": {"topic": "index_safety", "question_type": "factoid", "difficulty": "hard"}, "verification": {"verdict": "reanchored_at_gate", "reason": "Original anchor (graph-extractor two-LLM-calls text) was deleted with the graph module (roadmap 2.5). Re-anchored 2026-08-09 at the validation gate to the embedder-identity-guard prose in system_overview.md; topic dimension graph->index_safety. Recorded so pre/post comparisons account for it."}} +{"id": "docs_d11", "corpus": "docs", "query": "What confidence value is interpreted as a failed parse rather than a real score?", "expected": ["0 is treated as a parse failure"], "match": "any", "fact_ids": ["docs_verifier_zero_score"], "answer": "A confidence score of 0 is treated as a parse failure and no tag is appended.", "dimensions": {"topic": "verifier", "question_type": "factoid", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "tightened so the answer is the value, not the behaviour", "generated_query": "What value indicates that a tag could not be appended due to parsing failure?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d12", "corpus": "docs", "query": "How do I change the embedding model?", "expected": ["Changing the embedding model requires re-indexing"], "match": "any", "fact_ids": ["docs_reindex_required"], "answer": "Changing the embedding model requires re-indexing.", "dimensions": {"topic": "embedding_model", "question_type": "procedural", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How do I change the embedding model?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d13", "corpus": "docs", "query": "how do I enable pruning in localGPT?", "expected": ["so pruning is off unless a request enables it"], "match": "any", "fact_ids": ["docs_pruning_off_by_default"], "answer": "No shipped profile has a provence block, so pruning is off by default.", "dimensions": {"topic": "pruning", "question_type": "procedural", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "how do I enable pruning in localGPT?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d14", "corpus": "docs", "query": "Why do documents indexed from the command line end up with a different chunk size than ones indexed through the HTTP API?", "expected": ["while the HTTP path always sends"], "match": "any", "fact_ids": ["docs_chunk_size_layering"], "answer": "The CLI path uses the 1500-token code default while the HTTP path always sends 512.", "dimensions": {"topic": "chunking", "question_type": "procedural", "difficulty": "hard"}, "verification": {"verdict": "rescued", "reason": "model returned a truncated, off-topic {\"question\": ...} payload", "generated_query": "{\"question\":\"If you are using the HTTP path to generate code instead of the CLI option, how does this alter your output token limit?\"}", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d15", "corpus": "docs", "query": "How are plain-text uploads processed differently from PDFs and Word files?", "expected": ["files bypass docling entirely and are wrapped in a fenced code block"], "match": "any", "fact_ids": ["docs_txt_bypasses_docling"], "answer": ".txt files bypass docling and are wrapped in a fenced code block.", "dimensions": {"topic": "conversion", "question_type": "procedural", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "generated question presupposed a 'standard processing pipeline' the docs do not name", "generated_query": "How can I handle plain text files if they do not go through the standard processing pipeline?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d16", "corpus": "docs", "query": "Does this project build an approximate-nearest-neighbour index, and what does that mean for how a vector query executes?", "expected": ["No ANN index is created", "Vector search is a brute-force scan"], "match": "all", "fact_ids": ["docs_no_ann_index", "docs_brute_force_vector"], "answer": "No ANN/IVF-PQ index is created anywhere in rag_system. No ANN index is built, so vector search is an exhaustive scan.", "dimensions": {"topic": "vector_index", "question_type": "comparative", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question contained both expected strings", "generated_query": "Is vector search in rag_system a brute-force scan because no ANN index is created?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d17", "corpus": "docs", "query": "Which model handles routing and verification, and how much extra work does verification add per query?", "expected": ["costs an LLM call, on the utility model"], "match": "all", "fact_ids": ["docs_verifier_cost", "docs_triage_utility_model"], "answer": "Verification costs one extra utility-model round-trip per answered query. Both routers run on the utility model, not the generation model.", "dimensions": {"topic": "verifier", "question_type": "comparative", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "generated question was about swapping models, which the anchors do not cover | expected string re-anchored 2026-08-09 after gateway-router removal rewrote the source sentence.", "generated_query": "How does switching to a different large language model for both verification and routing stages impact the total inference rounds required per query?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d18", "corpus": "docs", "query": "Does the indexer overlap chunks or parallelise the work across workers?", "expected": ["the legacy path has no overlap logic at all", "there is no thread or process pool anywhere in the indexing path"], "match": "all", "fact_ids": ["docs_no_overlap_knob", "docs_indexing_sequential"], "answer": "There is no chunk-overlap knob; the legacy chunker has no overlap logic. Indexing is sequential; batch size controls reporting and memory, not concurrency.", "dimensions": {"topic": "chunking", "question_type": "comparative", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "generated question asked about 'distributed systems features', not answerable from the anchors", "generated_query": "In what fundamental ways do the legacy chunker and its indexing pipeline lack modern architectural features found in distributed systems?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d19", "corpus": "docs", "query": "Can I tune how much the keyword leg counts versus the vector leg when the two are combined?", "expected": ["There is no weighted linear blend"], "match": "any", "fact_ids": ["docs_no_weighted_blend"], "answer": "Hybrid fusion is pure RRF; there is no weighted linear blend and no dense_weight knob.", "dimensions": {"topic": "hybrid_retrieval", "question_type": "negative", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question contained the expected string 'weighted linear blend'", "generated_query": "Is there a weighted linear blend option available in the localGPT project?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d20", "corpus": "docs", "query": "Does the generated answer contain markers pointing at the passage each claim came from?", "expected": ["There are no inline citation markers"], "match": "any", "fact_ids": ["docs_no_citation_markers"], "answer": "There are no inline citation markers; sources come back as source_documents.", "dimensions": {"topic": "synthesis", "question_type": "negative", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question contained the expected string 'inline citation marker'", "generated_query": "Is there an inline citation marker in localGPT?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d21", "corpus": "docs", "query": "Does the agent do any pattern matching on the query before it asks a model to route it?", "expected": ["There is no regex or keyword stage in the agent"], "match": "any", "fact_ids": ["docs_triage_no_regex"], "answer": "The agent router has no regex or keyword stage.", "dimensions": {"topic": "triage", "question_type": "negative", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question contained the expected string 'regex or keyword stage'", "generated_query": "Is there a regex or keyword stage in the agent router?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d22", "corpus": "docs", "query": "can I completely disable the triage mechanism in this system?", "expected": ["There is no global triage on/off switch"], "match": "any", "fact_ids": ["docs_triage_no_switch"], "answer": "There is no global triage on/off switch and no similarity threshold.", "dimensions": {"topic": "triage", "question_type": "negative", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "can I completely disable the triage mechanism in this system?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d23", "corpus": "docs", "query": "Are there instances where contexts retrieved as part of neighbor expansion are subsequently removed?", "expected": ["the freshly added neighbours are filtered back out"], "match": "any", "fact_ids": ["docs_expansion_filtered_out"], "answer": "When the reranker ran, context-expansion neighbours are filtered back out.", "dimensions": {"topic": "context_expansion", "question_type": "negative", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "Are there instances where contexts retrieved as part of neighbor expansion are subsequently removed?", "generator_model": "qwen3.5:4b"}} +{"id": "docs_d24", "corpus": "docs", "query": "Why does a vector-dimension mismatch raise an error instead of just rebuilding the table?", "expected": ["Silently dropping or recreating an index would corrupt it"], "match": "any", "fact_ids": ["docs_dimension_mismatch_raises"], "answer": "A dimension mismatch is a hard ValueError on purpose.", "dimensions": {"topic": "vector_index", "question_type": "negative", "difficulty": "hard"}, "verification": {"verdict": "rescued", "reason": "model returned {\"question\": ...}", "generated_query": "{\"question\":\"What prevents an index from being safely recreated when its dimensions do not match?\"}", "generator_model": "qwen3.5:4b"}} diff --git a/eval/goldset/hr.jsonl b/eval/goldset/hr.jsonl new file mode 100644 index 00000000..8cf228b5 --- /dev/null +++ b/eval/goldset/hr.jsonl @@ -0,0 +1,24 @@ +{"id": "hr_h01", "corpus": "hr", "query": "How many days of paid annual leave do employees below Grade 7 accrue per calendar year?", "expected": ["23 days of paid annual leave"], "match": "any", "fact_ids": ["hr_annual_below_g7"], "answer": "Employees below Grade 7 accrue 23 days of paid annual leave per calendar year.", "dimensions": {"topic": "annual_leave", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many days of paid annual leave do employees below Grade 7 accrue per calendar year?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h02", "corpus": "hr", "query": "How many days do Grade 7 and above accrue?", "expected": ["Grade 7 and above accrue 28 days"], "match": "any", "fact_ids": ["hr_annual_g7_plus"], "answer": "Grade 7 and above accrue 28 days.", "dimensions": {"topic": "annual_leave", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many days do Grade 7 and above accrue?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h03", "corpus": "hr", "query": "At what salary percentage is sick leave paid for the first 12 weeks?", "expected": ["100 percent of base salary for the first 12 weeks"], "match": "any", "fact_ids": ["hr_sick_full_pay"], "answer": "Sick leave is paid at 100 percent for the first 12 weeks.", "dimensions": {"topic": "sick_leave", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "At what salary percentage is sick leave paid for the first 12 weeks?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h04", "corpus": "hr", "query": "How many weeks of parental leave per child are fully paid?", "expected": ["Parental leave is 18 weeks per child, of which 6 weeks are fully paid"], "match": "any", "fact_ids": ["hr_parental_length"], "answer": "18 weeks per child, 6 of them fully paid.", "dimensions": {"topic": "parental_leave", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many weeks of parental leave per child are fully paid?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h05", "corpus": "hr", "query": "How many days of leave are available for an immediate family member?", "expected": ["5 working days for an immediate family member"], "match": "any", "fact_ids": ["hr_bereavement"], "answer": "5 days immediate family, 2 days extended family.", "dimensions": {"topic": "bereavement", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many days of leave are available for an immediate family member?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h06", "corpus": "hr", "query": "How many working days is jury service paid in full for per year?", "expected": ["paid in full for up to 15 working days"], "match": "any", "fact_ids": ["hr_jury_duty"], "answer": "Jury service is paid in full up to 15 working days per year.", "dimensions": {"topic": "jury_duty", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many working days is jury service paid in full for per year?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h07", "corpus": "hr", "query": "How many public holidays does Northwind Robotics recognise?", "expected": ["recognises 9 public holidays"], "match": "any", "fact_ids": ["hr_public_holiday_count"], "answer": "Nine recognised public holidays.", "dimensions": {"topic": "public_holidays", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How many public holidays does Northwind Robotics recognise?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h08", "corpus": "hr", "query": "What is the identifier and revision number of the leave policy?", "expected": ["Policy PPL-204, revision 4"], "match": "any", "fact_ids": ["hr_policy_id"], "answer": "The policy is PPL-204 revision 4, effective 1 February 2026.", "dimensions": {"topic": "policy_metadata", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question quoted 'PPL-204 revision 4', i.e. the expected string", "generated_query": "What is the effective date of policy PPL-204 revision 4?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h09", "corpus": "hr", "query": "Which department owns this policy, and where is it based?", "expected": ["Department of People Operations, Northwind Robotics, Gothenburg"], "match": "any", "fact_ids": ["hr_policy_owner"], "answer": "Owned by People Operations in Gothenburg.", "dimensions": {"topic": "policy_metadata", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question asked who owns the department, inverting the fact", "generated_query": "Who owns the Department of People Operations at Northwind Robotics?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h10", "corpus": "hr", "query": "What is the maximum duration of an unpaid sabbatical?", "expected": ["unpaid sabbatical of up to 90 days"], "match": "any", "fact_ids": ["hr_sabbatical_length"], "answer": "Unpaid sabbatical of up to 90 days.", "dimensions": {"topic": "sabbatical", "question_type": "factoid", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What is the maximum duration of an unpaid sabbatical?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h11", "corpus": "hr", "query": "What date marks the end of the expiration period for rollover leave days?", "expected": ["Carried days expire on 31 March"], "match": "any", "fact_ids": ["hr_carryover_expiry"], "answer": "Carried days expire on 31 March and are not paid out.", "dimensions": {"topic": "annual_leave", "question_type": "factoid", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What date marks the end of the expiration period for rollover leave days?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h12", "corpus": "hr", "query": "After the initial full-pay period of a long illness ends, what proportion of salary continues and for how long?", "expected": ["at 60 percent for a further 8 weeks"], "match": "any", "fact_ids": ["hr_sick_reduced_pay"], "answer": "Then 60 percent for a further 8 weeks.", "dimensions": {"topic": "sick_leave", "question_type": "factoid", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "generated question invented a 'one month / next two months' schedule the policy does not state", "generated_query": "If a person takes partial leave for one month and continues after that period, what percentage rate applies to the next two months?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h13", "corpus": "hr", "query": "What is the multiplier applied to a standard wage when an employee works through a public holiday and gets paid?", "expected": ["paid at 1.5 times the normal rate"], "match": "any", "fact_ids": ["hr_public_holiday_pay"], "answer": "Working a public holiday pays 1.5x plus a substitute day.", "dimensions": {"topic": "public_holidays", "question_type": "factoid", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What is the multiplier applied to a standard wage when an employee works through a public holiday and gets paid?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h14", "corpus": "hr", "query": "How far in advance do I have to file a leave request, and where do I file it?", "expected": ["Kestrel HR portal at least 10"], "match": "any", "fact_ids": ["hr_request_notice"], "answer": "Requests go through the Kestrel HR portal at least 10 working days ahead.", "dimensions": {"topic": "requesting_leave", "question_type": "procedural", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question contained 'Kestrel HR portal', part of the expected string", "generated_query": "How many working days ahead must requests go through the Kestrel HR portal?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h15", "corpus": "hr", "query": "At what point during a sickness absence do I have to produce a doctor's note?", "expected": ["exceeds 4 consecutive working days"], "match": "any", "fact_ids": ["hr_medical_certificate"], "answer": "A medical certificate is required past 4 consecutive working days.", "dimensions": {"topic": "sick_leave", "question_type": "procedural", "difficulty": "easy"}, "verification": {"verdict": "rewritten", "reason": "generated question contained '4 consecutive working days', the expected string", "generated_query": "What happens if I need to take more than 4 consecutive working days off due to a medical issue?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h16", "corpus": "hr", "query": "What do I need to qualify for an extended unpaid break, how much notice must I give, and who signs it off?", "expected": ["at least 4 years of continuous service", "require 60 days written", "approved by the Head of People Operations"], "match": "all", "fact_ids": ["hr_sabbatical_eligibility", "hr_sabbatical_notice", "hr_sabbatical_approver"], "answer": "Sabbatical eligibility requires 4 years of continuous service. Sabbatical applications require 60 days written notice. Approved by the Head of People Operations.", "dimensions": {"topic": "sabbatical", "question_type": "procedural", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "'a leave away from work' was too vague to be answerable by the sabbatical section specifically", "generated_query": "What conditions must be met and what steps are needed to take a leave away from work?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h17", "corpus": "hr", "query": "How can an employee obtain authorization for a run of more than ten back-to-back workdays?", "expected": ["consecutive working days additionally requires written approval"], "match": "any", "fact_ids": ["hr_director_approval"], "answer": "Absences over 10 consecutive working days need written director approval.", "dimensions": {"topic": "requesting_leave", "question_type": "procedural", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How can an employee obtain authorization for a run of more than ten back-to-back workdays?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h18", "corpus": "hr", "query": "How does annual leave entitlement differ between employees below Grade 7 and those at Grade 7 or above?", "expected": ["23 days of paid annual leave", "Grade 7 and above accrue 28 days"], "match": "all", "fact_ids": ["hr_annual_below_g7", "hr_annual_g7_plus"], "answer": "Employees below Grade 7 accrue 23 days of paid annual leave per calendar year. Grade 7 and above accrue 28 days.", "dimensions": {"topic": "annual_leave", "question_type": "comparative", "difficulty": "easy"}, "verification": {"verdict": "rescued", "reason": "model returned a truncated {\"question\": ...} payload", "generated_query": "{\"question\":\"What is the difference in paid annual leave accrual between an employee below Grade 7 and one at that grade?\"}", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h19", "corpus": "hr", "query": "How does sick pay in the first three months of an absence compare with the months that follow?", "expected": ["100 percent of base salary for the first 12 weeks", "at 60 percent for a further 8 weeks"], "match": "all", "fact_ids": ["hr_sick_full_pay", "hr_sick_reduced_pay"], "answer": "Sick leave is paid at 100 percent for the first 12 weeks. Then 60 percent for a further 8 weeks.", "dimensions": {"topic": "sick_leave", "question_type": "comparative", "difficulty": "hard"}, "verification": {"verdict": "rewritten", "reason": "generated question referenced 'after 20 weeks', which is past the end of the stated schedule", "generated_query": "How does the paid sick leave percentage for employees in their second month of absence compare to that after 20 weeks?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h20", "corpus": "hr", "query": "How do unused annual leave days behave regarding both the maximum amount that can be rolled over and their expiration date?", "expected": ["maximum of 5 unused annual leave days may be carried", "Carried days expire on 31 March"], "match": "all", "fact_ids": ["hr_carryover_cap", "hr_carryover_expiry"], "answer": "At most 5 unused annual leave days carry into the next year. Carried days expire on 31 March and are not paid out.", "dimensions": {"topic": "annual_leave", "question_type": "comparative", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "How do unused annual leave days behave regarding both the maximum amount that can be rolled over and their expiration date?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h21", "corpus": "hr", "query": "Are agency contractors covered by the leave and absence policy?", "expected": ["Contractors engaged through an agency"], "match": "any", "fact_ids": ["hr_contractors_excluded"], "answer": "Agency contractors are not covered by the policy.", "dimensions": {"topic": "exclusions", "question_type": "negative", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "Are agency contractors covered by the leave and absence policy?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h22", "corpus": "hr", "query": "What happens to my annual leave if I resign before completing six months of service?", "expected": ["not paid out on resignation"], "match": "any", "fact_ids": ["hr_resignation_payout"], "answer": "Annual leave is not paid out on resignation under 6 months of service.", "dimensions": {"topic": "exclusions", "question_type": "negative", "difficulty": "easy"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "What happens to my annual leave if I resign before completing six months of service?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h23", "corpus": "hr", "query": "Can a parent utilize their leave entitlement after the child reaches three years of age?", "expected": ["before the child's third birthday"], "match": "any", "fact_ids": ["hr_parental_deadline"], "answer": "Parental leave must be taken before the child's third birthday.", "dimensions": {"topic": "parental_leave", "question_type": "negative", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "Can a parent utilize their leave entitlement after the child reaches three years of age?", "generator_model": "qwen3.5:4b"}} +{"id": "hr_h24", "corpus": "hr", "query": "Can a user schedule more than three distinct parental leave segments without violating the rules?", "expected": ["no more than 3 separate blocks"], "match": "any", "fact_ids": ["hr_parental_blocks"], "answer": "Parental leave may be split into at most 3 blocks.", "dimensions": {"topic": "parental_leave", "question_type": "negative", "difficulty": "hard"}, "verification": {"verdict": "accepted", "reason": null, "generated_query": "Can a user schedule more than three distinct parental leave segments without violating the rules?", "generator_model": "qwen3.5:4b"}} diff --git a/eval/goldset/multiturn.jsonl b/eval/goldset/multiturn.jsonl new file mode 100644 index 00000000..30d08826 --- /dev/null +++ b/eval/goldset/multiturn.jsonl @@ -0,0 +1,12 @@ +{"id": "mt_01", "corpus": "hr", "class": "answer-entity", "turns": ["What is the identifier and revision number of the Northwind leave policy?", "When did that revision take effect?"], "expected": ["1 February 2026"], "match": "any", "answer": "Revision 4 of policy PPL-204 is effective 1 February 2026.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "hr chunk 0: 'Policy PPL-204, revision 4. Effective 1 February 2026.'"}} +{"id": "mt_02", "corpus": "hr", "class": "user-turn", "turns": ["How many weeks of parental leave per child does the policy provide?", "Can it be split into separate blocks?", "And by when must all of it be taken?"], "expected": ["third birthday"], "match": "any", "answer": "Parental leave must be taken before the child's third birthday.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "hr chunk 0: 'It must be taken before the child's third birthday.'"}} +{"id": "mt_03", "corpus": "hr", "class": "answer-entity", "turns": ["What happens if I am rostered to work on a public holiday?", "By when do I have to use that substitute day?"], "expected": ["same quarter"], "match": "any", "answer": "The substitute day off must be taken within the same quarter.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "hr chunk 1: 'paid at 1.5 times the normal rate and receives a substitute day off within the same quarter.'"}} +{"id": "mt_04", "corpus": "atlas7", "class": "answer-entity", "turns": ["Who makes the Atlas-7 espresso machine, and where are they based?", "How long is the parts warranty they provide?"], "expected": ["36-month", "36 month"], "match": "any", "answer": "Meridian Coffee Systems provides a 36-month parts warranty.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "atlas7 chunk 0: 'The Atlas-7 carries a 36-month parts warranty.'"}} +{"id": "mt_05", "corpus": "atlas7", "class": "user-turn", "turns": ["What temperature does the PID hold the brew water at?", "How tight is the tolerance on that?"], "expected": ["0.4 degrees"], "match": "any", "answer": "The brew water temperature is held to a tolerance of 0.4 degrees.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "atlas7 chunk 0: '93.5 degrees Celsius with a tolerance of 0.4 degrees.'"}} +{"id": "mt_06", "corpus": "atlas7", "class": "answer-entity", "turns": ["How do I clear the E42 error on the Atlas-7?", "What is that pump rated at?"], "expected": ["52 watts"], "match": "any", "answer": "The vibratory pump is rated for 52 watts continuous duty.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "atlas7 chunk 0: E42 is pump cavitation (pump named only in the answer); 'The vibratory pump is rated for 52 watts continuous duty.'"}} +{"id": "mt_07", "corpus": "acq", "class": "answer-entity", "turns": ["Which single customer accounts for the largest share of StartupXYZ's revenue?", "Has their consent to the change of control been obtained, and when was it received?"], "expected": ["February 10, 2025"], "match": "any", "answer": "Yes - MegaCorp's consent was obtained; it was received on February 10, 2025.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "02_due_diligence: 'Largest customer (MegaCorp) accounts for 28%'; 09_customer_consents: 'MegaCorp Inc. - OBTAINED ... Consent Received: February 10, 2025'."}} +{"id": "mt_08", "corpus": "acq", "class": "answer-entity", "turns": ["What is the one pending intellectual-property matter at StartupXYZ?", "How does the risk memo rate it?"], "expected": ["LOW", "minor risk"], "match": "any", "answer": "The pending patent application (No. 17/456,789) is rated LOW / a minor risk item in the Risk Assessment Memo.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "03_ip_certification: 'one pending patent application (Application No. 17/456,789) ... noted in Risk Assessment Memo as a minor risk item'; 04_risk_assessment: '3.1 Pending Patent Application (LOW)'."}} +{"id": "mt_09", "corpus": "acq", "class": "user-turn", "turns": ["How much did the parties pay to file under Hart-Scott-Rodino?", "When was that filing made?", "And when was early termination of the waiting period granted?"], "expected": ["January 28, 2025"], "match": "any", "answer": "Early termination was granted on January 28, 2025.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "08_regulatory_approval: 'January 28, 2025 ... Early Termination of HSR Waiting Period'; 10_closing_checklist: 'Early termination received (January 28, 2025)'."}} +{"id": "mt_10", "corpus": "docs", "class": "user-turn", "turns": ["What model does localGPT use for sentence pruning?", "Is it enabled by default?"], "expected": ["off unless a request enables it", "disabled by default", "off by default", "not enabled by default"], "match": "any", "answer": "No - pruning is off unless a request enables it.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "docs gold d01/d13: 'naver/provence-reranker-debertav3-v1'; 'pruning is off unless a request enables it'."}} +{"id": "mt_11", "corpus": "rfc", "class": "answer-entity", "turns": ["What formula does RFC 9002 give for computing the probe timeout period?", "What value does the spec recommend for the granularity constant in that formula?"], "expected": ["1 millisecond", "1 ms"], "match": "any", "answer": "The recommended value of kGranularity is 1 millisecond.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "RFC 9002: 'PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay'; 'The RECOMMENDED value of the timer granularity (kGranularity) is 1 millisecond.'"}} +{"id": "mt_12", "corpus": "rfc", "class": "answer-entity", "turns": ["What version number appears in the long header of QUIC version 2 packets?", "How was that value generated?"], "expected": ["sha256", "first four bytes"], "match": "any", "answer": "0x6b3343cf was generated by taking the first four bytes of the sha256sum of 'QUICv2 version number'.", "verification": {"author": "hand-authored 2026-08-15", "provenance": "RFC 9369: 'The Version field of long headers is 0x6b3343cf. This was generated by taking the first four bytes of the sha256sum of \"QUICv2 version number\".'"}} diff --git a/eval/goldset/paraphrases.jsonl b/eval/goldset/paraphrases.jsonl new file mode 100644 index 00000000..4bbe2f6c --- /dev/null +++ b/eval/goldset/paraphrases.jsonl @@ -0,0 +1,120 @@ +{"id": "rfc_q01", "original": "If a QUIC endpoint never sends the ack_delay_exponent transport parameter, what value is its peer supposed to assume?", "para": "When one side of a QUIC connection omits the setting that tells its partner how much to scale reported acknowledgement-delay values, what number should the receiving side fall back on by default?"} +{"id": "rfc_q02", "original": "Is a QUIC endpoint free to advertise any value it likes for active_connection_id_limit, or is there a floor?", "para": "When a QUIC endpoint states how many connection identifiers it will keep active at once, can it announce an arbitrarily small number, or is there a minimum it must satisfy?"} +{"id": "rfc_q03", "original": "Which QUIC transport error code does an endpoint use once it concludes that the network path can no longer carry its traffic?", "para": "When a QUIC participant decides its current network route is no longer able to carry its data, which standardized transport error code is it supposed to report?"} +{"id": "rfc_q04", "original": "What round-trip time estimate should a QUIC sender start from before it has measured any samples of its own?", "para": "Before it has gathered any actual timing measurements of its own, what starting estimate for the network round trip should a QUIC data sender assume?"} +{"id": "rfc_q05", "original": "When the probe timeout fires in QUIC, how many probe datagrams is the sender permitted to put on the wire?", "para": "Once the timer meant to catch a possibly-stalled QUIC connection expires, how many exploratory packets is the sending side allowed to transmit?"} +{"id": "rfc_q06", "original": "Which frame type codepoints does the unreliable datagram extension add to QUIC?", "para": "What numeric identifiers get introduced for the message type added by the QUIC add-on that lets data be carried without delivery guarantees?"} +{"id": "rfc_q07", "original": "What version number appears in the long header of QUIC version 2 packets?", "para": "In packets belonging to the second version of the QUIC protocol, what value shows up in the version field of the extended packet header?"} +{"id": "rfc_q08", "original": "Leaving any specific QUIC version aside, what range of lengths may the Destination Connection ID field take?", "para": "Regardless of which particular revision of the protocol is in use, what is the permissible size range, in bytes, for the field that identifies the receiving side's connection?"} +{"id": "rfc_q09", "original": "May a DNS-over-QUIC client open its connection on UDP port 53?", "para": "Is it acceptable for a client doing domain-name lookups over QUIC to establish its session on the traditional DNS UDP port, port 53?"} +{"id": "rfc_q10", "original": "How much flow-control credit should an HTTP/3 endpoint hand to each unidirectional stream its peer may open?", "para": "For each one-way data channel that a partner is permitted to open, how much data-sending allowance should an HTTP-over-QUIC participant grant it initially?"} +{"id": "rfc_q11", "original": "If a peer never sends the QPACK dynamic table capacity setting, what capacity does the other side assume?", "para": "When the other side of a connection never communicates the parameter governing how large its adjustable header-compression table may grow, what size should be assumed by default?"} +{"id": "rfc_q12", "original": "Which header fields, if present, make an HTTP message ineligible for the Capsule Protocol?", "para": "Which set of HTTP header fields, when included in a message, rule that message out for use with the mechanism for wrapping datagram-like payloads inside an HTTP request or response body?"} +{"id": "rfc_q13", "original": "In the extensible priority scheme for HTTP, what urgency applies to a request that does not state one, and what is the permitted range?", "para": "Under the flexible request-importance framework for HTTP, what is the full span of allowed priority levels, and which one applies automatically when a request leaves it unspecified?"} +{"id": "rfc_q14", "original": "Whereabouts in a QUIC short header does an on-path observer find the latency spin bit?", "para": "Within the abbreviated packet header format QUIC uses, at exactly which bit position can someone monitoring traffic along the path locate the indicator used for round-trip latency measurement?"} +{"id": "rfc_q15", "original": "The QUIC loss recovery document says a server's probe packets are still bound by the anti-amplification limit from the transport specification. How much may a server send to an address it has not yet validated?", "para": "According to the document covering how QUIC handles lost packets, a server's exploratory retransmissions still fall under a cap\u2014defined in the core transport rules\u2014on how much unsolicited data may go to a destination that hasn't yet been confirmed as reachable. What is that ceiling, expressed relative to what the server has already received from that destination?"} +{"id": "rfc_q16", "original": "RFC 9000 tells a client to send another Initial packet when the probe timeout defined in the recovery specification expires. How is that timer actually computed?", "para": "The core QUIC transport rules direct a client to transmit another first-flight packet once the retransmission-detection timer set out in the loss-and-congestion document runs out. What formula is used to calculate how long that timer runs?"} +{"id": "rfc_q17", "original": "The QUIC transport specification hands the Retry packet's Integrity Tag off to the QUIC-TLS document. How wide is that tag?", "para": "The main QUIC transport rules leave the sizing of the authentication tag attached to a Retry packet to the document describing how TLS is used to secure QUIC. How many bits long is that tag?"} +{"id": "rfc_q18", "original": "QUIC version 2 replaces the Initial-keys salt that Section 5.2 of the QUIC-TLS document specifies. What are the two salts \u2014 the one version 2 defines and the original it replaces?", "para": "The second version of QUIC swaps out the fixed starting value used to derive its earliest encryption keys, replacing the one laid out in the document describing how TLS protects the original QUIC. What are those two starting-value constants \u2014 the new one and the one it supersedes?"} +{"id": "rfc_q19", "original": "RFC 9001 says a QUIC endpoint that cannot agree on an application protocol closes the connection with a no_application_protocol TLS alert. What numeric alert description does TLS assign to that alert?", "para": "When two sides of a QUIC connection cannot settle on a shared application-layer protocol, the connection gets torn down using a specific TLS alert reserved for that failure. What numeric code does TLS assign to that alert?"} +{"id": "rfc_q20", "original": "Permanent registrations in the QUIC registries created by RFC 9000 use the Specification Required policy. What does that policy demand over and above a designated expert's review?", "para": "Lasting entries added to the IANA registries set up by the core QUIC transport document must follow a particular approval policy. Beyond having a qualified reviewer sign off, what extra requirement does that policy impose?"} +{"id": "rfc_q21", "original": "RFC 9220 bootstraps WebSockets over HTTP/3 with Extended CONNECT. Where was the pseudo-header it depends on first defined, and on what sort of request may it appear?", "para": "The document describing how to run WebSockets on top of HTTP/3 relies on an extended form of the CONNECT method. The special pseudo-header that mechanism needs was originally introduced elsewhere \u2014 where was it first specified, and what kind of request is it allowed to appear on?"} +{"id": "rfc_q22", "original": "The HTTP/3 ORIGIN extension says its payload semantics are identical to the HTTP/2 frame it is based on. What frame type number does that HTTP/2 frame carry, and what does its payload hold?", "para": "An add-on for HTTP/3 that lets a server declare which other origins it also speaks for states that its message contents behave exactly like those of the equivalent HTTP/2 frame. What numeric type code identifies that HTTP/2 frame, and what does it carry inside?"} +{"id": "rfc_q23", "original": "Connect-UDP tunnels carry their data stream with the Capsule Protocol. Which capsule type does that protocol define for carrying datagrams?", "para": "Tunnels that proxy UDP traffic over HTTP wrap their payload using the mechanism for embedding datagram-like capsules inside an HTTP message body. Which specific capsule kind does that mechanism designate for holding the actual datagram data?"} +{"id": "rfc_q24", "original": "HTTP/3 insists that each peer permit at least three unidirectional streams. Which QUIC transport parameter expresses that limit, and what is its identifier?", "para": "HTTP-over-QUIC requires that each side of a connection allow the other to open at least three one-way data channels. Which underlying QUIC connection-setup setting conveys that cap, and what numeric identifier does it carry?"} +{"id": "acq_q01", "original": "What is the total purchase price for the StartupXYZ acquisition?", "para": "How much did the acquiring company agree to pay in total to buy the target company?"} +{"id": "acq_q02", "original": "When was the mutual non-disclosure agreement between TechCorp and StartupXYZ signed?", "para": "On what date did the acquiring company and the target company execute their mutual confidentiality agreement?"} +{"id": "acq_q03", "original": "How many United States patents does StartupXYZ own?", "para": "How many patents registered in the United States belong to the target company?"} +{"id": "acq_q04", "original": "What proportion of the target company's turnover comes from its single biggest client?", "para": "What share of the seller's total sales revenue does its top customer generate?"} +{"id": "acq_q05", "original": "Which categories of information are carved out of the confidentiality obligations in the NDA?", "para": "What kinds of data are excluded from the confidentiality duties in the non-disclosure agreement?"} +{"id": "acq_q06", "original": "What must each side do with the other's confidential material once it is asked for or the arrangement ends?", "para": "When the deal wraps up or either side requests it, what are both parties obligated to do with the other's sensitive information?"} +{"id": "acq_q07", "original": "How much did the parties pay to file under Hart-Scott-Rodino?", "para": "What was the cost of submitting the required antitrust pre-merger notification?"} +{"id": "acq_q08", "original": "Which subject did the seller's counsel refuse to give a view on in its opinion letter?", "para": "In its formal legal opinion, what topic did the seller's attorneys decline to address?"} +{"id": "acq_q09", "original": "How does the borrowing the reviewers first disclosed compare with the extra borrowing found later, and what was the later amount?", "para": "How does the debt initially reported during due diligence compare to the additional debt discovered afterward, and how large was that later amount?"} +{"id": "acq_q10", "original": "How much annual revenue was flagged as being at risk from the largest customer, and did that customer end up agreeing to the change of control?", "para": "What yearly revenue was identified as being in jeopardy from the biggest customer, and did that customer ultimately approve the ownership change?"} +{"id": "acq_q11", "original": "The finance team revised the cash portion of the deal. What figure did they land on, and does the closing paperwork carry the same number?", "para": "After the cash component of the transaction was adjusted, what was the new figure, and does it match the amount in the closing documents?"} +{"id": "acq_q12", "original": "How large is the target's workforce and how is it split between functions?", "para": "What is the total headcount at the company being acquired, and how is it broken down by department?"} +{"id": "acq_q13", "original": "Article IV of the Acquisition Agreement makes the buyer's obligation to close conditional on a regulatory approval. On what date was that condition satisfied?", "para": "The purchase agreement makes the buyer's closing duty contingent on obtaining government clearance. When was that requirement fulfilled?"} +{"id": "acq_q14", "original": "Employee matters under the Acquisition Agreement are pushed to Schedule 3. What was the estimated price tag of the retention packages contemplated there?", "para": "The purchase agreement defers staffing issues to a separate attachment. What was the projected cost of the retention incentives described there?"} +{"id": "acq_q15", "original": "The Acquisition Agreement fixes the price by reference to Exhibit A - Financial Terms. After the recommended write-downs, what did the price become?", "para": "The purchase agreement ties the deal value to a financial terms attachment. Once the suggested downward adjustments were made, what did the price come to?"} +{"id": "acq_q16", "original": "In the Acquisition Agreement the seller warrants that its intellectual property is unencumbered, as confirmed by an outside certification. What did that certification conclude about liens on the patents?", "para": "The purchase agreement contains a seller promise that its IP is free of claims, backed by an independent certificate. What did that certificate say about liens on the patents?"} +{"id": "acq_q17", "original": "The NDA treats the target's financial information as confidential and points to where that information sits. What top-line revenue did that material report for FY2024?", "para": "The confidentiality agreement marks the company's financial figures as protected and references where they are located. What total revenue did that source show for fiscal year 2024?"} +{"id": "acq_q18", "original": "The risk memo rates a single pending patent application as a low-priority item. What is the serial number of that application?", "para": "The risk assessment flags one still-pending patent filing as a minor concern. What is that filing's application number?"} +{"id": "acq_q19", "original": "The legal opinion excepts change-of-control provisions from its no-conflicts view. For each affected customer, what is the current consent status?", "para": "The attorney's opinion carves out ownership-change clauses from its conflict-free conclusion. For each customer touched by those clauses, what is the present approval status?"} +{"id": "acq_q20", "original": "The due diligence report says a Hart-Scott-Rodino filing is needed and refers the timeline elsewhere. On what date was the filing submitted?", "para": "The diligence report notes that an antitrust pre-merger notification is required and points elsewhere for the schedule. When was that notification actually filed?"} +{"id": "acq_q21", "original": "The closing checklist calls for an escrow agreement tied to Exhibit C. Over what period is that escrow released?", "para": "The closing checklist requires a holdback arrangement described in an attachment. Over what span of time are those held-back funds paid out?"} +{"id": "acq_q22", "original": "The FTC's early-termination letter tells the parties they may proceed to closing. Where is that closing scheduled to take place?", "para": "The regulator's notice granting early clearance gives the parties the go-ahead to finalize the deal. At what location is the closing set to occur?"} +{"id": "acq_q23", "original": "The Acquisition Agreement sets out a cash payment and makes closing conditional on antitrust clearance. What is the revised cash figure, and when did the clearance come through?", "para": "The purchase agreement specifies a cash payment and requires antitrust approval before closing can happen. What was the updated cash amount, and when was that approval granted?"} +{"id": "acq_q24", "original": "What does early termination of the waiting period not prevent the Commission from doing?", "para": "Ending the waiting period ahead of schedule does not stop the regulatory agency from doing what?"} +{"id": "atlas7_a01", "original": "What is the pressure of the brew boiler during extraction?", "para": "While a shot is being pulled, how much pressure builds up in the coffee-side boiler?"} +{"id": "atlas7_a02", "original": "What pressure is the steam boiler held at?", "para": "At what bar reading is the boiler that generates steam kept steady?"} +{"id": "atlas7_a03", "original": "What temperature does the PID hold the brew water at?", "para": "To what temperature does the electronic temperature controller keep the brewing water?"} +{"id": "atlas7_a04", "original": "What is the continuous duty rating of the vibratory pump?", "para": "How many watts of continuous power can the vibration-style pump handle?"} +{"id": "atlas7_a05", "original": "What is the water hardness threshold for this machine?", "para": "What's the maximum water-hardness level this espresso maker can tolerate?"} +{"id": "atlas7_a06", "original": "What part number is the group head gasket?", "para": "What's the catalog number for the gasket fitted on the group head?"} +{"id": "atlas7_a07", "original": "What is the duration of the parts warranty?", "para": "How long does the coverage period for replacement parts last?"} +{"id": "atlas7_a08", "original": "Who makes the Atlas-7, and where are they based?", "para": "Which company manufactures the Atlas-7, and where is it headquartered?"} +{"id": "atlas7_a09", "original": "What is the revision of the Atlas-7 Dual Boiler espresso machine?", "para": "What revision label applies to the Atlas-7 Dual Boiler unit?"} +{"id": "atlas7_a10", "original": "Where is the serial number located?", "para": "Where on the unit can the serial number be found?"} +{"id": "atlas7_a11", "original": "How tightly is the brew water temperature controlled?", "para": "How much variance is allowed in the brewing water's temperature?"} +{"id": "atlas7_a12", "original": "Which component should be swapped out to clear the E11 fault code?", "para": "To resolve fault code E11, which part must be replaced?"} +{"id": "atlas7_a13", "original": "How do I fix E42 pump cavitation on the espresso machine?", "para": "What's the remedy when the espresso maker displays an E42 pump-cavitation alert?"} +{"id": "atlas7_a14", "original": "How often should I backflush the espresso machine?", "para": "At what interval should backflushing be performed on the espresso maker?"} +{"id": "atlas7_a15", "original": "How often should I run the descaling cycle if my water contains more than 120 ppm of hardness?", "para": "If my water tests above 120 parts per million in hardness, how often should the descaling cycle be run?"} +{"id": "atlas7_a16", "original": "The machine is not registering any water flow at all - what should I check?", "para": "What should be checked if the unit registers zero water flow whatsoever?"} +{"id": "atlas7_a17", "original": "What should an operator do to address steam overpressure?", "para": "What steps should the user take to resolve excessive steam pressure?"} +{"id": "atlas7_a18", "original": "what is the pressure difference between the brew boiler and steam boiler during extraction?", "para": "During brewing, how do the coffee boiler's pressure and the steam boiler's pressure compare?"} +{"id": "atlas7_a19", "original": "Which needs doing more often on this machine: descaling, or replacing the group head gasket?", "para": "Between descaling and swapping the group head gasket, which chore comes around more often on this unit?"} +{"id": "atlas7_a20", "original": "Which error code points to a failed temperature sensor, and which one points to excess steam pressure?", "para": "Which fault code signals a broken temperature probe, and which one signals too much steam pressure?"} +{"id": "atlas7_a21", "original": "What kind of descaling product will void the warranty?", "para": "Which category of descaling agent will invalidate the warranty?"} +{"id": "atlas7_a22", "original": "Will a warranty claim be accepted without a serial number, and where do I find it?", "para": "Is a serial number required for a warranty claim to be honored, and where would I locate it?"} +{"id": "atlas7_a23", "original": "Is there a limit on how far the brew temperature may drift before it is out of spec?", "para": "Is there a ceiling on how much the brew temperature can deviate before it's out of spec?"} +{"id": "atlas7_a24", "original": "under what pressure should the OPV calibration be performed if steam overpressure is detected?", "para": "When a steam-overpressure condition occurs, at what bar setting should the overpressure valve be calibrated?"} +{"id": "hr_h01", "original": "How many days of paid annual leave do employees below Grade 7 accrue per calendar year?", "para": "For staff ranked under Grade 7, how many paid vacation days pile up over the course of a year?"} +{"id": "hr_h02", "original": "How many days do Grade 7 and above accrue?", "para": "How many days does someone at Grade 7 or higher rank earn?"} +{"id": "hr_h03", "original": "At what salary percentage is sick leave paid for the first 12 weeks?", "para": "What portion of your regular pay do you receive during the initial 12 weeks of being off sick?"} +{"id": "hr_h04", "original": "How many weeks of parental leave per child are fully paid?", "para": "For each child, how many weeks of new-parent leave come with full pay?"} +{"id": "hr_h05", "original": "How many days of leave are available for an immediate family member?", "para": "If a close relative in your immediate family dies, how many days can you take off work?"} +{"id": "hr_h06", "original": "How many working days is jury service paid in full for per year?", "para": "If called for jury duty, for how many working days each year will you keep getting full pay?"} +{"id": "hr_h07", "original": "How many public holidays does Northwind Robotics recognise?", "para": "How many official holidays does Northwind Robotics observe?"} +{"id": "hr_h08", "original": "What is the identifier and revision number of the leave policy?", "para": "What code number and version number identify the time-off policy?"} +{"id": "hr_h09", "original": "Which department owns this policy, and where is it based?", "para": "Which team is responsible for this policy, and where are they located?"} +{"id": "hr_h10", "original": "What is the maximum duration of an unpaid sabbatical?", "para": "What's the longest an unpaid, extended leave of absence from work can last?"} +{"id": "hr_h11", "original": "What date marks the end of the expiration period for rollover leave days?", "para": "By what calendar date do vacation days carried forward from last year become void?"} +{"id": "hr_h12", "original": "After the initial full-pay period of a long illness ends, what proportion of salary continues and for how long?", "para": "Once the period of getting full salary during an extended illness comes to an end, what percentage of pay kicks in next, and how many weeks does it last?"} +{"id": "hr_h13", "original": "What is the multiplier applied to a standard wage when an employee works through a public holiday and gets paid?", "para": "By what factor is regular pay multiplied for someone who works on an official holiday?"} +{"id": "hr_h14", "original": "How far in advance do I have to file a leave request, and where do I file it?", "para": "How much lead time is needed before requesting time off, and what system do you use to submit it?"} +{"id": "hr_h15", "original": "At what point during a sickness absence do I have to produce a doctor's note?", "para": "After how much time being off work sick does someone need to hand in a note from their doctor?"} +{"id": "hr_h16", "original": "What do I need to qualify for an extended unpaid break, how much notice must I give, and who signs it off?", "para": "What are the requirements to become eligible for a long stretch of unpaid time off, how far ahead do I need to give notice, and whose approval is required?"} +{"id": "hr_h17", "original": "How can an employee obtain authorization for a run of more than ten back-to-back workdays?", "para": "What does staff need to do to get sign-off for being away from work more than ten consecutive working days in a row?"} +{"id": "hr_h18", "original": "How does annual leave entitlement differ between employees below Grade 7 and those at Grade 7 or above?", "para": "How does the amount of paid vacation time earned differ for staff ranked below Grade 7 compared with those at Grade 7 or higher?"} +{"id": "hr_h19", "original": "How does sick pay in the first three months of an absence compare with the months that follow?", "para": "During an extended sick leave, how does the percentage of pay someone receives in roughly the first three months compare to the following months?"} +{"id": "hr_h20", "original": "How do unused annual leave days behave regarding both the maximum amount that can be rolled over and their expiration date?", "para": "For vacation days that go unused, what's the cap on how many can be carried into the next year, and by when do they run out?"} +{"id": "hr_h21", "original": "Are agency contractors covered by the leave and absence policy?", "para": "Does the time-off and absence policy apply to contractors hired through a staffing agency?"} +{"id": "hr_h22", "original": "What happens to my annual leave if I resign before completing six months of service?", "para": "If I quit my job before I've worked here for six months, what becomes of my unused vacation days?"} +{"id": "hr_h23", "original": "Can a parent utilize their leave entitlement after the child reaches three years of age?", "para": "Is it possible for a parent to take their allotted time off for a new child once that child has turned three years old?"} +{"id": "hr_h24", "original": "Can a user schedule more than three distinct parental leave segments without violating the rules?", "para": "Is an employee permitted to split their parental leave into more than three separate periods and still be within policy?"} +{"id": "docs_d01", "original": "What model is used for sentence pruning?", "para": "Which underlying model handles trimming irrelevant sentences out of retrieved passages?"} +{"id": "docs_d02", "original": "How many characters does the verifier prompt clamp context to?", "para": "When assembling the verifier's prompt, what character count does it cap the included context at?"} +{"id": "docs_d03", "original": "How many overviews does the overview router use?", "para": "At most how many document summaries does the overview-based routing step pull in?"} +{"id": "docs_d04", "original": "How many dimensions does the Qwen3-Embedding-4B produce?", "para": "What's the length of the output vector produced by the Qwen3-Embedding-4B embedding model?"} +{"id": "docs_d05", "original": "How many characters is the overview input truncated to?", "para": "What's the character limit the text gets cut down to before it's used to build an overview?"} +{"id": "docs_d06", "original": "What is the maximum number of sentences in a reply generated by direct_answer?", "para": "What's the upper limit on sentence count for a response coming out of the direct_answer prompt?"} +{"id": "docs_d07", "original": "How many extra vectors does turning on late chunking write?", "para": "If late chunking is switched on, roughly how many additional vectors wind up getting written?"} +{"id": "docs_d08", "original": "What does the enricher do when the model returns an almost-empty summary?", "para": "How does the enrichment step react when the model hands back a summary that's practically blank?"} +{"id": "docs_d09", "original": "How large is the per-retriever cache that stores previously embedded queries?", "para": "What's the capacity of the cache, maintained separately per retriever, that holds query embeddings computed earlier?"} +{"id": "docs_d10", "original": "What stops two embedding models with the same vector width from silently corrupting an index?", "para": "What mechanism keeps two distinct embedding models that happen to share an output size from quietly corrupting an index?"} +{"id": "docs_d11", "original": "What confidence value is interpreted as a failed parse rather than a real score?", "para": "Which specific confidence number gets read as a parsing failure rather than treated as a genuine score?"} +{"id": "docs_d12", "original": "How do I change the embedding model?", "para": "What's involved in switching over to a different embedding model?"} +{"id": "docs_d13", "original": "how do I enable pruning in localGPT?", "para": "What steps turn context pruning on in this project?"} +{"id": "docs_d14", "original": "Why do documents indexed from the command line end up with a different chunk size than ones indexed through the HTTP API?", "para": "Why does indexing a document via the CLI end up producing a different chunk size than indexing the same kind of document through the REST/HTTP interface?"} +{"id": "docs_d15", "original": "How are plain-text uploads processed differently from PDFs and Word files?", "para": "In what way does handling of raw .txt file uploads diverge from how PDFs and Word documents get handled?"} +{"id": "docs_d16", "original": "Does this project build an approximate-nearest-neighbour index, and what does that mean for how a vector query executes?", "para": "Is an approximate nearest-neighbor style index constructed here, and given that, how does a vector similarity lookup actually get carried out?"} +{"id": "docs_d17", "original": "Which model handles routing and verification, and how much extra work does verification add per query?", "para": "Which model takes care of both directing queries and checking the resulting answers, and what additional overhead does that checking step add per query?"} +{"id": "docs_d18", "original": "Does the indexer overlap chunks or parallelise the work across workers?", "para": "When documents get indexed, do the resulting chunks overlap, and is that workload split across multiple worker threads or processes?"} +{"id": "docs_d19", "original": "Can I tune how much the keyword leg counts versus the vector leg when the two are combined?", "para": "Is there a setting to adjust the relative weight given to the keyword-based results versus the vector-based results when merging them?"} +{"id": "docs_d20", "original": "Does the generated answer contain markers pointing at the passage each claim came from?", "para": "Do the answers this system produces include inline markers linking each claim back to its source passage?"} +{"id": "docs_d21", "original": "Does the agent do any pattern matching on the query before it asks a model to route it?", "para": "Before the query gets handed to a model for routing, does the agent run any rule-based or keyword matching on it first?"} +{"id": "docs_d22", "original": "can I completely disable the triage mechanism in this system?", "para": "Is it possible to fully switch off the query-triage feature in this system?"} +{"id": "docs_d23", "original": "Are there instances where contexts retrieved as part of neighbor expansion are subsequently removed?", "para": "Can chunks pulled in via neighboring-context expansion later get filtered back out?"} +{"id": "docs_d24", "original": "Why does a vector-dimension mismatch raise an error instead of just rebuilding the table?", "para": "Why does a mismatch in vector dimensionality trigger a hard exception instead of the system just rebuilding the table on its own?"} diff --git a/eval/goldset/rfc.jsonl b/eval/goldset/rfc.jsonl new file mode 100644 index 00000000..f8a51cd9 --- /dev/null +++ b/eval/goldset/rfc.jsonl @@ -0,0 +1,24 @@ +{"id": "rfc_q01", "corpus": "rfc", "query": "If a QUIC endpoint never sends the ack_delay_exponent transport parameter, what value is its peer supposed to assume?", "expected": ["default value of 3 is assumed (indicating a multiplier of 8)."], "match": "any", "fact_ids": ["q9000_ack_delay_exponent_default"], "answer": "A default of 3 is assumed, which means ACK Delay values are scaled by a multiplier of 8.", "expected_sources": ["RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt"], "anchor_doc": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", "multi_document": false, "dimensions": {"topic": "transport_parameters", "question_type": "factoid", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q02", "corpus": "rfc", "query": "Is a QUIC endpoint free to advertise any value it likes for active_connection_id_limit, or is there a floor?", "expected": ["active_connection_id_limit parameter MUST be at least 2."], "match": "any", "fact_ids": ["q9000_active_cid_limit_floor"], "answer": "There is a floor: the advertised active_connection_id_limit must be at least 2.", "expected_sources": ["RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt"], "anchor_doc": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", "multi_document": false, "dimensions": {"topic": "transport_parameters", "question_type": "negative", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q03", "corpus": "rfc", "query": "Which QUIC transport error code does an endpoint use once it concludes that the network path can no longer carry its traffic?", "expected": ["NO_VIABLE_PATH (0x10): An endpoint has determined that the network"], "match": "any", "fact_ids": ["q9000_no_viable_path"], "answer": "NO_VIABLE_PATH, error code 0x10.", "expected_sources": ["RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt"], "anchor_doc": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", "multi_document": false, "dimensions": {"topic": "error_codes", "question_type": "factoid", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q04", "corpus": "rfc", "query": "What round-trip time estimate should a QUIC sender start from before it has measured any samples of its own?", "expected": ["SHOULD be set to 333 milliseconds. This results in handshakes"], "match": "any", "fact_ids": ["q9002_initial_rtt"], "answer": "333 milliseconds \u2014 the recommended kInitialRtt.", "expected_sources": ["RFC 9002 - QUIC Loss Detection and Congestion Control.txt"], "anchor_doc": "RFC 9002 - QUIC Loss Detection and Congestion Control.txt", "multi_document": false, "dimensions": {"topic": "loss_recovery", "question_type": "factoid", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q05", "corpus": "rfc", "query": "When the probe timeout fires in QUIC, how many probe datagrams is the sender permitted to put on the wire?", "expected": ["MAY send up to two full-sized datagrams containing ack-eliciting"], "match": "any", "fact_ids": ["q9002_pto_probe_count"], "answer": "At least one ack-eliciting packet, and it may send up to two full-sized datagrams.", "expected_sources": ["RFC 9002 - QUIC Loss Detection and Congestion Control.txt"], "anchor_doc": "RFC 9002 - QUIC Loss Detection and Congestion Control.txt", "multi_document": false, "dimensions": {"topic": "loss_recovery", "question_type": "procedural", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q06", "corpus": "rfc", "query": "Which frame type codepoints does the unreliable datagram extension add to QUIC?", "expected": ["form 0b0011000X (or the values 0x30 and 0x31). The least significant"], "match": "any", "fact_ids": ["q9221_datagram_frame_types"], "answer": "DATAGRAM frames use the two values 0x30 and 0x31 (0b0011000X); the low bit is the LEN bit.", "expected_sources": ["RFC 9221 - An Unreliable Datagram Extension to QUIC.txt"], "anchor_doc": "RFC 9221 - An Unreliable Datagram Extension to QUIC.txt", "multi_document": false, "dimensions": {"topic": "datagram_extension", "question_type": "factoid", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q07", "corpus": "rfc", "query": "What version number appears in the long header of QUIC version 2 packets?", "expected": ["The Version field of long headers is 0x6b3343cf. This was generated"], "match": "any", "fact_ids": ["q9369_version_number"], "answer": "0x6b3343cf.", "expected_sources": ["RFC 9369 - QUIC Version 2.txt"], "anchor_doc": "RFC 9369 - QUIC Version 2.txt", "multi_document": false, "dimensions": {"topic": "version_negotiation", "question_type": "factoid", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q08", "corpus": "rfc", "query": "Leaving any specific QUIC version aside, what range of lengths may the Destination Connection ID field take?", "expected": ["the Destination Connection ID Length field and is between 0 and 255"], "match": "any", "fact_ids": ["q8999_cid_length_range"], "answer": "Between 0 and 255 bytes \u2014 the version-independent invariant, which is wider than version 1's own 20-byte cap.", "expected_sources": ["RFC 8999 - Version-Independent Properties of QUIC.txt"], "anchor_doc": "RFC 8999 - Version-Independent Properties of QUIC.txt", "multi_document": false, "dimensions": {"topic": "invariants", "question_type": "factoid", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q09", "corpus": "rfc", "query": "May a DNS-over-QUIC client open its connection on UDP port 53?", "expected": ["DoQ connections MUST NOT use UDP port 53."], "match": "any", "fact_ids": ["q9250_no_port_53"], "answer": "No. DoQ uses UDP port 853 by default and port 53 is explicitly forbidden.", "expected_sources": ["RFC 9250 - DNS over Dedicated QUIC Connections.txt"], "anchor_doc": "RFC 9250 - DNS over Dedicated QUIC Connections.txt", "multi_document": false, "dimensions": {"topic": "transport", "question_type": "negative", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q10", "corpus": "rfc", "query": "How much flow-control credit should an HTTP/3 endpoint hand to each unidirectional stream its peer may open?", "expected": ["provide at least 1,024 bytes of flow-control credit to each"], "match": "any", "fact_ids": ["q9114_uni_stream_credit"], "answer": "At least 1,024 bytes of flow-control credit per unidirectional stream.", "expected_sources": ["RFC 9114 - HTTP3.txt"], "anchor_doc": "RFC 9114 - HTTP3.txt", "multi_document": false, "dimensions": {"topic": "stream_setup", "question_type": "procedural", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q11", "corpus": "rfc", "query": "If a peer never sends the QPACK dynamic table capacity setting, what capacity does the other side assume?", "expected": ["SETTINGS_QPACK_MAX_TABLE_CAPACITY (0x01): The default value is zero."], "match": "any", "fact_ids": ["q9204_qpack_capacity_default"], "answer": "Zero \u2014 SETTINGS_QPACK_MAX_TABLE_CAPACITY (0x01) defaults to zero, so no dynamic table entries may be inserted.", "expected_sources": ["RFC 9204 - QPACK Field Compression for HTTP3.txt"], "anchor_doc": "RFC 9204 - QPACK Field Compression for HTTP3.txt", "multi_document": false, "dimensions": {"topic": "settings", "question_type": "factoid", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q12", "corpus": "rfc", "query": "Which header fields, if present, make an HTTP message ineligible for the Capsule Protocol?", "expected": ["The Capsule Protocol MUST NOT be used with messages that contain Content-Length, Content-Type, or Transfer-Encoding header fields."], "match": "any", "fact_ids": ["q9297_capsule_forbidden_headers"], "answer": "Content-Length, Content-Type and Transfer-Encoding.", "expected_sources": ["RFC 9297 - HTTP Datagrams and the Capsule Protocol.txt"], "anchor_doc": "RFC 9297 - HTTP Datagrams and the Capsule Protocol.txt", "multi_document": false, "dimensions": {"topic": "capsule_protocol", "question_type": "negative", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q13", "corpus": "rfc", "query": "In the extensible priority scheme for HTTP, what urgency applies to a request that does not state one, and what is the permitted range?", "expected": ["between 0 and 7 inclusive, in descending order of priority. The default is 3."], "match": "any", "fact_ids": ["q9218_urgency_default"], "answer": "Urgency runs from 0 to 7 in descending order of priority and defaults to 3.", "expected_sources": ["RFC 9218 - Extensible Prioritization Scheme for HTTP.txt"], "anchor_doc": "RFC 9218 - Extensible Prioritization Scheme for HTTP.txt", "multi_document": false, "dimensions": {"topic": "priority_parameters", "question_type": "factoid", "difficulty": "easy", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q14", "corpus": "rfc", "query": "Whereabouts in a QUIC short header does an on-path observer find the latency spin bit?", "expected": ["latency spin bit: The third-most-significant bit of the first octet"], "match": "any", "fact_ids": ["q9312_spin_bit_position"], "answer": "It is the third-most-significant bit of the first octet of the short header.", "expected_sources": ["RFC 9312 - Manageability of the QUIC Transport Protocol.txt"], "anchor_doc": "RFC 9312 - Manageability of the QUIC Transport Protocol.txt", "multi_document": false, "dimensions": {"topic": "observability", "question_type": "factoid", "difficulty": "hard", "requires_crossref": false}, "verification": {"verdict": "accepted", "reason": null, "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q15", "corpus": "rfc", "query": "The QUIC loss recovery document says a server's probe packets are still bound by the anti-amplification limit from the transport specification. How much may a server send to an address it has not yet validated?", "expected": ["to the unvalidated address to three times the amount of data received"], "match": "any", "fact_ids": ["q9000_anti_amplification"], "answer": "No more than three times the amount of data it has received from that address.", "expected_sources": ["RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt"], "anchor_doc": "RFC 9002 - QUIC Loss Detection and Congestion Control.txt", "multi_document": false, "dimensions": {"topic": "address_validation", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "premise is RFC 9002's deferral; the limit itself is only stated in RFC 9000 Section 8.1", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q16", "corpus": "rfc", "query": "RFC 9000 tells a client to send another Initial packet when the probe timeout defined in the recovery specification expires. How is that timer actually computed?", "expected": ["PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay"], "match": "any", "fact_ids": ["q9002_pto_formula"], "answer": "PTO = smoothed_rtt + max(4*rttvar, kGranularity) + max_ack_delay.", "expected_sources": ["RFC 9002 - QUIC Loss Detection and Congestion Control.txt"], "anchor_doc": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", "multi_document": false, "dimensions": {"topic": "loss_recovery", "question_type": "factoid", "difficulty": "easy", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "RFC 9000 only points at Section 6.2 of RFC 9002; the formula lives there", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q17", "corpus": "rfc", "query": "The QUIC transport specification hands the Retry packet's Integrity Tag off to the QUIC-TLS document. How wide is that tag?", "expected": ["The Retry Integrity Tag is a 128-bit field that is computed as the"], "match": "any", "fact_ids": ["q9001_retry_integrity_tag"], "answer": "128 bits, computed as the output of AEAD_AES_128_GCM over the Retry pseudo-packet.", "expected_sources": ["RFC 9001 - Using TLS to Secure QUIC.txt"], "anchor_doc": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", "multi_document": false, "dimensions": {"topic": "packet_protection", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "RFC 9000 references Section 5.8 of RFC 9001; the width is stated only in RFC 9001", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q18", "corpus": "rfc", "query": "QUIC version 2 replaces the Initial-keys salt that Section 5.2 of the QUIC-TLS document specifies. What are the two salts \u2014 the one version 2 defines and the original it replaces?", "expected": ["initial_salt = 0x0dede3def700a6db819381be6e269dcbf9bd2ed9", "initial_salt = 0x38762cf7f55934b34d179ae6a4c80cadccbb7f0a"], "match": "all", "fact_ids": ["q9369_initial_salt_v2", "q9001_initial_salt_v1"], "answer": "Version 2 uses 0x0dede3def700a6db819381be6e269dcbf9bd2ed9; version 1's salt, in RFC 9001, is 0x38762cf7f55934b34d179ae6a4c80cadccbb7f0a.", "expected_sources": ["RFC 9369 - QUIC Version 2.txt", "RFC 9001 - Using TLS to Secure QUIC.txt"], "anchor_doc": "RFC 9369 - QUIC Version 2.txt", "multi_document": true, "dimensions": {"topic": "key_derivation", "question_type": "comparative", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "multi-document: one salt in each of RFC 9369 and RFC 9001", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q19", "corpus": "rfc", "query": "RFC 9001 says a QUIC endpoint that cannot agree on an application protocol closes the connection with a no_application_protocol TLS alert. What numeric alert description does TLS assign to that alert?", "expected": ["no_application_protocol(120),"], "match": "any", "fact_ids": ["q7301_alpn_alert_value"], "answer": "120.", "expected_sources": ["RFC 7301 - TLS Application-Layer Protocol Negotiation Extension.txt"], "anchor_doc": "RFC 9001 - Using TLS to Secure QUIC.txt", "multi_document": false, "dimensions": {"topic": "alpn", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "RFC 9001 names the alert; its numeric value is defined only in RFC 7301", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q20", "corpus": "rfc", "query": "Permanent registrations in the QUIC registries created by RFC 9000 use the Specification Required policy. What does that policy demand over and above a designated expert's review?", "expected": ["This policy is the same as Expert Review, with the additional requirement of a formal public specification."], "match": "any", "fact_ids": ["q8126_specification_required"], "answer": "A formal, permanent and readily available public specification \u2014 Specification Required is Expert Review plus that document.", "expected_sources": ["RFC 8126 - Guidelines for Writing an IANA Considerations Section in RFCs.txt"], "anchor_doc": "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt", "multi_document": false, "dimensions": {"topic": "registration_policies", "question_type": "procedural", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "RFC 9000 names the policy and cites RFC 8126; the definition is in RFC 8126", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q21", "corpus": "rfc", "query": "RFC 9220 bootstraps WebSockets over HTTP/3 with Extended CONNECT. Where was the pseudo-header it depends on first defined, and on what sort of request may it appear?", "expected": ["A new pseudo-header field :protocol MAY be included on request"], "match": "any", "fact_ids": ["q8441_protocol_pseudo_header"], "answer": "In RFC 8441: a new pseudo-header field :protocol may be included on request HEADERS carrying the CONNECT method.", "expected_sources": ["RFC 8441 - Bootstrapping WebSockets with HTTP2.txt"], "anchor_doc": "RFC 9220 - Bootstrapping WebSockets with HTTP3.txt", "multi_document": false, "dimensions": {"topic": "extended_connect", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "RFC 9220 reuses Extended CONNECT without restating it; the definition is in RFC 8441", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q22", "corpus": "rfc", "query": "The HTTP/3 ORIGIN extension says its payload semantics are identical to the HTTP/2 frame it is based on. What frame type number does that HTTP/2 frame carry, and what does its payload hold?", "expected": ["The ORIGIN frame type is 0xc (decimal 12) and contains zero or more"], "match": "any", "fact_ids": ["q8336_origin_frame_type"], "answer": "The HTTP/2 ORIGIN frame is type 0xc (decimal 12) and carries zero or more Origin-Entry blocks.", "expected_sources": ["RFC 8336 - The ORIGIN HTTP2 Frame.txt"], "anchor_doc": "RFC 9412 - The ORIGIN Extension in HTTP3.txt", "multi_document": false, "dimensions": {"topic": "origin_frame", "question_type": "factoid", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "RFC 9412 defers payload semantics to RFC 8336", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q23", "corpus": "rfc", "query": "Connect-UDP tunnels carry their data stream with the Capsule Protocol. Which capsule type does that protocol define for carrying datagrams?", "expected": ["This document defines the DATAGRAM (0x00) Capsule Type."], "match": "any", "fact_ids": ["q9297_datagram_capsule_type"], "answer": "The DATAGRAM Capsule Type, value 0x00.", "expected_sources": ["RFC 9297 - HTTP Datagrams and the Capsule Protocol.txt"], "anchor_doc": "RFC 9298 - Proxying UDP in HTTP.txt", "multi_document": false, "dimensions": {"topic": "capsule_protocol", "question_type": "factoid", "difficulty": "easy", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "RFC 9298 points at Section 3.2 of RFC 9297; the capsule type is defined in RFC 9297", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} +{"id": "rfc_q24", "corpus": "rfc", "query": "HTTP/3 insists that each peer permit at least three unidirectional streams. Which QUIC transport parameter expresses that limit, and what is its identifier?", "expected": ["Therefore, the transport parameters sent by both clients and servers MUST allow the peer to create at least three", "initial_max_streams_uni (0x09): The initial maximum unidirectional"], "match": "all", "fact_ids": ["q9114_three_uni_streams", "q9000_initial_max_streams_uni"], "answer": "initial_max_streams_uni, transport parameter 0x09, defined in RFC 9000; HTTP/3 requires it to allow at least three unidirectional streams.", "expected_sources": ["RFC 9114 - HTTP3.txt", "RFC 9000 - QUIC A UDP-Based Multiplexed and Secure Transport.txt"], "anchor_doc": "RFC 9114 - HTTP3.txt", "multi_document": true, "dimensions": {"topic": "stream_setup", "question_type": "comparative", "difficulty": "hard", "requires_crossref": true}, "verification": {"verdict": "accepted", "reason": "multi-document: the requirement is in RFC 9114, the parameter it names is defined in RFC 9000", "generated_query": null, "generator_model": null, "author": "hand-authored 2026-08-12", "provenance": "hand-authored 2026-08-12"}} diff --git a/eval/judge.py b/eval/judge.py new file mode 100644 index 00000000..2d23683b --- /dev/null +++ b/eval/judge.py @@ -0,0 +1,327 @@ +"""Binary groundedness judge (Phase 0.3). + +Binary pass/fail, not a Likert score, per the practitioner canon the roadmap +cites. Given (question, answer, evidence chunks) the judge returns:: + + {"grounded": true | false, "reason": "<short justification>"} + +It runs on the utility model (``qwen3.5:4b`` by default) through the repo's own +``OllamaClient``, which sends ``think: false`` alongside ``format="json"`` โ€” +without that, a thinking model puts its JSON in the ``thinking`` field and +returns an empty ``response``. + +A model name starting with ``claude-`` routes to the Anthropic API instead +(eval-only โ€” the product stays fully local; requires ``pip install anthropic`` +and credentials in the environment). Motivation: the Phase-4 A/Bs showed the +4b judge returning verdicts its own reasons contradict on exactly the rows +that decide feature adoption (``eval/decisions/phase4-escalation-rerun.md`` +ยง6). Select it per run with ``JUDGE_MODEL=claude-sonnet-5``. Both backends are +pinned to temperature 0 โ€” a judge must be deterministic to be comparable +across runs. + +Regardless of backend, the verifier's ``[Confidence: N%] [Warning: ...]`` +suffix is stripped from the ANSWER before judging โ€” one judge reason was +observed citing the confidence figure as grounds for rejection. + +Validate before trusting it:: + + .venv/bin/python eval/judge.py --validate + +That scores the judge against ``eval/judge_validation.jsonl`` (20 hand-built +cases, 10 grounded / 10 subtly ungrounded) and prints the confusion matrix, TPR +and TNR. The roadmap's gate is >=90% overall agreement. + +Judge a single case ad hoc:: + + .venv/bin/python eval/judge.py --question "..." --answer "..." --evidence "..." +""" + +import argparse +import json +import os +import re +import sys +from datetime import datetime, timezone + +EVAL_DIR = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, os.path.abspath(os.path.join(EVAL_DIR, ".."))) + +from rag_system.utils.ollama_client import OllamaClient # noqa: E402 + +VALIDATION_PATH = os.path.join(EVAL_DIR, "judge_validation.jsonl") +RESULTS_DIR = os.path.join(EVAL_DIR, "results") + +# Prompt versions are kept, not overwritten, so the history stays auditable. +# All three were run against judge_validation.jsonl on 2026-08-08 (numbers in +# eval/BASELINE.md): v1 20/20, v2 15/20 (TPR 0.50 โ€” the extra strictness makes +# it reject correct answers), v3 20/20. v1 is the default: it ties v3 on this +# set and is the shorter prompt. v2 is kept as the recorded counter-example. +PROMPTS = { + "v1": """You are a strict fact-checker. + +EVIDENCE: +{evidence} + +QUESTION: {question} + +ANSWER: {answer} + +Decide whether the ANSWER is fully supported by the EVIDENCE. + +Rules: +- Every factual claim in the ANSWER must appear in the EVIDENCE. +- If any number, name, part identifier, duration or condition differs from the + EVIDENCE, the answer is NOT grounded. +- If the ANSWER adds a claim the EVIDENCE does not state, it is NOT grounded, + even if that claim sounds plausible or is true in the real world. +- Do not use any outside knowledge. The EVIDENCE is the only truth. + +Respond with JSON only: {{"grounded": true or false, "reason": "<one sentence>"}} +""", + + "v2": """You are a strict fact-checker. You compare an ANSWER against the EVIDENCE +it is supposed to be based on, and you have no other source of truth. + +EVIDENCE: +{evidence} + +QUESTION: {question} + +ANSWER: {answer} + +Work through the ANSWER one claim at a time. A claim is any number, quantity, +duration, percentage, part identifier, model name, person, department, place, +condition, threshold or instruction. + +For each claim ask: does the EVIDENCE state exactly this? + +Mark grounded = false if ANY of the following is true: +- a number, code or identifier in the ANSWER differs from the EVIDENCE, even by + one digit or one transposed pair of digits; +- the ANSWER attaches a value to the wrong subject (for example it swaps two + quantities, or credits one component with another component's figure); +- the ANSWER states something the EVIDENCE does not state at all, however + reasonable it sounds; +- the ANSWER attributes a procedure or condition to the wrong item. + +Mark grounded = true only when every claim in the ANSWER is directly supported +by the EVIDENCE. Extra caution, brevity or hedging in the ANSWER is fine and +does not by itself make it ungrounded. + +Respond with JSON only: {{"grounded": true or false, "reason": "<one sentence naming the specific claim you checked>"}} +""", + + "v3": """You are a strict fact-checker. The EVIDENCE below is the only truth that +exists. Ignore everything you know about the world. + +EVIDENCE: +{evidence} + +QUESTION: {question} + +ANSWER: {answer} + +Task: decide whether every claim in the ANSWER is supported by the EVIDENCE. + +Procedure: +1. List, silently, every number, code, identifier, name, duration, percentage, + threshold and instruction that appears in the ANSWER. +2. For each one, locate it in the EVIDENCE, character by character for numbers + and part codes. +3. If one of them is absent, altered (9.2 vs 9.5, TS-71 vs TS-17, 60 vs 90), or + attached to a different subject than in the EVIDENCE, the ANSWER is NOT + grounded. +4. If the ANSWER contains an instruction, entitlement or consequence that the + EVIDENCE never states, the ANSWER is NOT grounded, no matter how plausible. +5. Otherwise the ANSWER is grounded. Paraphrasing, reordering, summarising and + omitting details are all fine โ€” only added or altered content is a failure. + +Respond with JSON only: +{{"grounded": true or false, "reason": "<one sentence naming the exact claim that decided it>"}} +""", +} + +DEFAULT_VERSION = "v1" +DEFAULT_MODEL = os.getenv("JUDGE_MODEL") or os.getenv("ENRICHMENT_MODEL") or "qwen3.5:4b" + + +def strip_think(text: str) -> str: + return re.sub(r"<think>.*?</think>", "", text or "", flags=re.S).strip() + + +# The agent's verifier appends this to every answer it checks. It is metadata +# about the answer, not part of it, and it measurably perturbs the judge. +_VERIFIER_SUFFIX = re.compile( + r"\s*\[Confidence:\s*\d+%\]\s*(\[Warning:[^\]]*\])?\s*$") + + +def strip_verifier_suffix(text: str) -> str: + return _VERIFIER_SUFFIX.sub("", text or "").strip() + + +class GroundednessJudge: + def __init__(self, model: str = DEFAULT_MODEL, host: str | None = None, + version: str = DEFAULT_VERSION): + if version not in PROMPTS: + raise ValueError(f"unknown prompt version {version!r}; have {sorted(PROMPTS)}") + self.model = model + self.version = version + self._use_anthropic = model.startswith("claude-") + if self._use_anthropic: + import anthropic # deferred so local-only runs don't need the package + self._anthropic = anthropic.Anthropic() + else: + self.client = OllamaClient(host=host or os.getenv("OLLAMA_HOST", "http://localhost:11434")) + + def _complete_anthropic(self, prompt: str) -> str: + """One judgment via the Anthropic API, JSON shape enforced server-side.""" + response = self._anthropic.messages.create( + model=self.model, + max_tokens=4096, # hard cap on thinking + response text together + # Same pin as the Ollama path below (sampling noise flipped 11/24 + # verdicts in the ftslc screen): a judge must be deterministic to + # be comparable across runs โ€” the API default is temperature 1.0. + temperature=0, + output_config={"format": {"type": "json_schema", "schema": { + "type": "object", + "properties": { + "grounded": {"type": "boolean"}, + "reason": {"type": "string"}, + }, + "required": ["grounded", "reason"], + "additionalProperties": False, + }}}, + messages=[{"role": "user", "content": prompt}], + ) + if response.stop_reason == "refusal": + return "" + return next((b.text for b in response.content if b.type == "text"), "") + + def judge(self, question: str, answer: str, evidence) -> dict: + if isinstance(evidence, (list, tuple)): + evidence_text = "\n\n---\n\n".join(str(e) for e in evidence) + else: + evidence_text = str(evidence) + + prompt = PROMPTS[self.version].format( + evidence=strip_verifier_suffix(evidence_text), question=question, + answer=strip_verifier_suffix(answer)) + if self._use_anthropic: + raw = self._complete_anthropic(prompt) + else: + # Temperature 0: sampling noise flipped 11/24 verdicts on + # byte-identical answers in the 2026-08-15 ftslc screen + # (eval/decisions/ftslc-index-fix-2026-08-15.md); a judge must be + # deterministic to be comparable across runs. + raw = strip_think((self.client.generate_completion( + model=self.model, prompt=prompt, format="json", + options={"temperature": 0}) or {}).get("response", "")) + + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + return {"grounded": None, "reason": "judge returned unparseable JSON", + "raw": raw, "error": "parse_error"} + + grounded = parsed.get("grounded") + if isinstance(grounded, str): + grounded = grounded.strip().lower() in ("true", "yes", "grounded") + if not isinstance(grounded, bool): + return {"grounded": None, "reason": "judge omitted a boolean 'grounded'", + "raw": raw, "error": "missing_field"} + return {"grounded": grounded, "reason": str(parsed.get("reason", "")).strip()} + + +def load_validation() -> list: + rows = [] + with open(VALIDATION_PATH, "r", encoding="utf-8") as fh: + for line in fh: + line = line.strip() + if line: + rows.append(json.loads(line)) + return sorted(rows, key=lambda r: r["id"]) + + +def validate(model: str, version: str, host: str | None) -> dict: + judge = GroundednessJudge(model=model, host=host, version=version) + cases = load_validation() + + tp = tn = fp = fn = errors = 0 + rows = [] + for case in cases: + verdict = judge.judge(case["question"], case["answer"], case["evidence"]) + predicted, label = verdict["grounded"], case["label_grounded"] + if predicted is None: + errors += 1 + outcome = "ERROR" + elif label and predicted: + tp += 1 + outcome = "TP" + elif label and not predicted: + fn += 1 + outcome = "FN" + elif not label and not predicted: + tn += 1 + outcome = "TN" + else: + fp += 1 + outcome = "FP" + rows.append({**case, "predicted": predicted, "outcome": outcome, + "judge_reason": verdict.get("reason"), "raw": verdict.get("raw")}) + flag = " " if outcome in ("TP", "TN") else "<" + print(f" {flag} {case['id']:<28} label={'grounded ' if label else 'UNgrounded'} " + f"pred={str(predicted):<5} {outcome:<5} {verdict.get('reason', '')[:70]}") + + positives, negatives = tp + fn, tn + fp + summary = { + "prompt_version": version, + "model": model, + "n": len(cases), + "confusion": {"TP": tp, "FN": fn, "TN": tn, "FP": fp, "unparseable": errors}, + "tpr": round(tp / positives, 4) if positives else None, + "tnr": round(tn / negatives, 4) if negatives else None, + "agreement": round((tp + tn) / len(cases), 4) if cases else None, + "timestamp": datetime.now(timezone.utc).isoformat(timespec="seconds"), + } + + print(f"\n prompt {version} on {model}") + print(f" confusion TP={tp} FN={fn} TN={tn} FP={fp} unparseable={errors}") + print(f" TPR (grounded correctly accepted) {summary['tpr']}") + print(f" TNR (ungrounded correctly rejected) {summary['tnr']}") + print(f" overall agreement {summary['agreement']} (gate: >= 0.90)") + + os.makedirs(RESULTS_DIR, exist_ok=True) + stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + out_path = os.path.join(RESULTS_DIR, f"judge_{version}_{stamp}.json") + with open(out_path, "w", encoding="utf-8") as fh: + json.dump({"summary": summary, "cases": rows}, fh, indent=2) + print(f" written {out_path}") + return summary + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--validate", action="store_true", help="score against judge_validation.jsonl") + parser.add_argument("--prompt-version", default=DEFAULT_VERSION, choices=sorted(PROMPTS)) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--host", default=None) + parser.add_argument("--question") + parser.add_argument("--answer") + parser.add_argument("--evidence", action="append", default=None) + args = parser.parse_args() + + if args.validate: + summary = validate(args.model, args.prompt_version, args.host) + return 0 if (summary["agreement"] or 0) >= 0.90 else 1 + + if not (args.question and args.answer and args.evidence): + parser.error("pass --validate, or all of --question / --answer / --evidence") + + judge = GroundednessJudge(model=args.model, host=args.host, version=args.prompt_version) + print(json.dumps(judge.judge(args.question, args.answer, args.evidence), indent=2)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/judge_hard_cases.jsonl b/eval/judge_hard_cases.jsonl new file mode 100644 index 00000000..cbc257a1 --- /dev/null +++ b/eval/judge_hard_cases.jsonl @@ -0,0 +1,18 @@ +{"id": "cell1_off_acq_q04", "question": "What proportion of the target company's turnover comes from its single biggest client?", "answer": "Based on the provided documents, the largest customer for StartupXYZ is **MegaCorp**, which accounts for **28%** of the company's revenue. MegaCorp represents a significant portion of startup companies and poses concentration risks that have been identified in due diligence reports (e.g., $3.4 million annual revenue at risk).\n\nThis high proportion has triggered several mitigations, including:\n- Obtaining change-of-control consent from MegaCorp prior to closing (which was obtained on February 10, 2025)\n- Establishing a contingent liability reserve of **$500,000** specifically for customer concentration risk in the Financial Adjustments Memo\n- Including customer concentration as a \"HIGH-PRIORITY RISK\" alongside key employee retention\n\nThe remaining customers (DataFlow and CloudTech combined with others) make up the other 72% of revenue. DataFlow Systems accounts for approximately **16%** ($1.2M annual), while CloudTech Partners represents about **9.5%** ($890K annual, pending consent as of February 15, 2025).\n\nThis customer concentration risk is documented across multiple reports:\n- Due Diligence Report (Section 4) noting \"Largest customer (MegaCorp) accounts for 28% of revenue - concentration risk noted\"\n- Risk Assessment Memo identifying this as a HIGH-priority concern with \"$3.4M annual revenue at risk\" if materialized [Confidence: 90%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "The largest customer, MegaCorp, accounts for 28% of revenue.", "label_grounded": true, "qwen4b_votes": 4, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_off_acq_q07", "question": "How much did the parties pay to file under Hart-Scott-Rodino?", "answer": "Answer:\nThe Federal Trade Commission (FTC) required an HSR Filing Fee of **$30,000** for TechCorp Industries' acquisition of StartupXYZ LLC. This information is explicitly stated in the context provided under \"FILING DETAILS\" within the Regulatory Approval Letter from the Premerger Notification Office dated January 28, 2025.\n\nThe document lists the following specific filing details:\n* **Transaction Value:** $45,000,000 (as stated in Document: Acquisition Agreement)\n* **HSR Filing Fee:** $30,000\n* **Filing Date:** January 10, 2025\n\nThe fee was paid as part of the Premerger Notification Office's process, which resulted in an early termination of the waiting period granted on January 28, 2025. The parties may now consummate the transaction at any time following this approval and subject to satisfaction of other conditions such as third-party consents and due diligence completion scheduled for closing on March 1, 2025.\n\nNote: This fee is distinct from the total acquisition purchase price of $45 million (later adjusted via Financial Adjustments Memo), which consists of cash payments ($30M initially, adjusted to ~$28.33M at closing), stock consideration ($10M), and earnout components ($5M). The HSR filing fee is a regulatory cost separate from the transaction's purchase price or financing arrangements totaling $39.6 million mentioned in the Closing Checklist. [Confidence: 100%]", "gold_answer": "The HSR filing fee was $30,000.", "label_grounded": true, "qwen4b_votes": 5, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_off_acq_q09", "question": "How does the borrowing the reviewers first disclosed compare with the extra borrowing found later, and what was the later amount?", "answer": "Answer:\nBased on the **Financial Adjustments Memo** dated December 23, 2024, regarding the StartupXYZ acquisition by TechCorp Industries, here are the details comparing previously disclosed debt with additional undisclosed borrowing found during due diligence:\n\n* **Previously Disclosed Debt**: The original financial records showed outstanding debt of **$1,500,000**.\n* **Extra Borrowing Found Later (Additional Identified Debt)**: A subsequent review identified an additional **$175,000** in previously undisclosed capital lease obligations.\n\nThe memo explicitly categorizes this new finding under the \"Debt Adjustment\" section and notes that it resulted in a corresponding adjustment of ($175,000) to the purchase price. The total original debt was therefore revised upward from $1.5 million to include these additional liabilities before being factored into the final adjusted purchase price calculation. [Confidence: 95%]", "gold_answer": "Due diligence disclosed $1.5 million of debt; a further $175,000 of capital lease obligations was identified afterwards.", "label_grounded": true, "qwen4b_votes": 4, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_off_acq_q12", "question": "How large is the target's workforce and how is it split between functions?", "answer": "Answer: I could not find that information in the provided documents. The retrieved snippets describe technical specifications for a Retrieval-Augmented Generation (RAG) system, including details about its architecture, chunking strategies, retrieval pipelines, embedding models, API endpoints, and database structures. They do not contain any information regarding the size of an organization's workforce or how that workforce is distributed across different functions. [Confidence: 100%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "47 employees: 32 in engineering, 8 in sales, 7 in operations.", "label_grounded": false, "qwen4b_votes": 0, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_off_docs_d03", "question": "How many overviews does the overview router use?", "answer": "Answer:\nBased on the provided documents, the number of document overviews used by the agent's triage router (specifically within `_route_via_overviews`) depends on which specific chunk is considered relevant for loading into memory versus what is conceptually available. The retrieval system explicitly states that when evaluating queries using the utility model in **layer 2** (`rag_system/agent/loop.py`), it builds a `DOCUMENT OVERVIEWS:` block from the first **40 loaded overviews**.\n\nThis detail appears repeatedly in the context:\n* One snippet specifies: \"Interpolates the loaded document overviews (first 40, :612-630 ) under a DOCUMENT OVERVIEWS: header\".\n* Another chunk notes that if no overview file exists for the session and there is no per-index file, `loop.py` returns None, meaning routing falls through to history short-circuits or fallback triage.\n\nHowever, other sections describe **indexing time** operations where overviews are generated. In this context:\n* The default behavior generates summaries using \"the first n chunks of each document (default 5)\".\n* These are written per index/session into `index_store/overviews/<session_or_index_id>.jsonl` or a global fallback file.\n\nTherefore, while **40 overviews** constitute the maximum window presented to the agent's routing model for decision-making during a query, the system initially creates summaries for only about 5 chunks per document by default before storing them in JSONL files (either per-session globally or as a session-specific override). The router reads these stored values up to its limit of using **40** at runtime if available. [Confidence: 95%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "The overview router uses the first 40 loaded overviews.", "label_grounded": true, "qwen4b_votes": 4, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_off_docs_d05", "question": "How many characters is the overview input truncated to?", "answer": "Answer: The document overview generation pipeline truncates its input prompt to **5000 characters**.\n\nThis specific constraint applies when generating overviews for documents; specifically, `OverviewBuilder` sends a request consisting of the first *first_n_chunks* chunks (where the default value is 5) truncated at this limit. This implementation detail ensures that large document summaries remain within reasonable length bounds before being passed to the enrichment model via `/api/generate`. The configuration also allows users to modify which utility or generation models are used for these tasks, with `qwen3.5:4b` typically serving as the default *enrichment_model* and `qwen3.5:9b` as the primary *generation_model*, though no prompt hard-codes specific model identities directlyโ€”resolving them dynamically instead. [Confidence: 100%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "Overview input is the first N chunks truncated to 5000 characters.", "label_grounded": true, "qwen4b_votes": 4, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_off_docs_d13", "question": "how do I enable pruning in localGPT?", "answer": "Answer: To enable sentence-level pruning (specifically via the **Provence** reranker, named `naver/provence-reranker-debertav3-v1`) in localGPT, you must explicitly toggle a feature flag. By default, this function is disabled (`provence.enabled` defaults to False) and should remain off to avoid unnecessary computational costs unless actively enabled by the user.\n\nBased on the system architecture and configuration details provided:\n\n**How it works:**\n* **Default State:** The \"Prune irrelevant sentences\" UI toggle corresponds to `provence_enabled`, which is set to false or disabled in the default pipeline profile (`default`). Even if an environment variable for a threshold (e.g., `providence_prune` on the API) is absent, it remains disabled.\n* **Activation:** You must enable pruning explicitly through one of these paths:\n * **User Interface:** Enable the toggle labeled **\"Prune irrelevant sentences\"** in the chat settings panel. This exposes the feature and activates the `provence_enabled` setting (typically via a profile change). The chunk references \"Stream phases\" being on by default, but mentions that pruning is an \"opt-in : provence.enabled defaults to False\".\n * **API Configuration:** On the RAG API endpoint (`POST /chat`), you can pass `provence_prune=true`. Note that this requires loading a DeBERTa model (approx. 30Mโ€“310M parameters depending on size) which is loaded lazily once behind a lock; if it fails to load, pruning skips rather than failing the query.\n* **Mechanism:** When active, after context expansion and before synthesis, Provence applies sentence labelling combined with reranking. Chunks deemed irrelevant are dropped (\"fully-pruned chunks dropped\"), preserving performance while reducing costโ€”provided that a cross-encoder (like BAAI/bge-reranker-v2-m3) is not already the default profile's primary method for context reduction.\n\n**Important Constraints:**\n* **Cross-dependency on Reranking:** Pruning works most effectively when reranking is enabled, as it folds into that stage at \"negligible to no drop in performance\". However, since localGPT defaults do not run a cross-encoder by default (specifically using the HuggingFace NLI/verifier or LLM-prompt verifier for grounding checks), Provence's argument of \"zero marginal cost\" does not hold unless reranking is also active.\n* **Re-indexing Not Required:** Unlike changing the embedding model which requires re-indexing, enabling pruning modifies how chunks are processed during retrieval and synthesis without altering the underlying vector store structure immediately (though it affects what gets returned).\n\n**Summary of Steps to Enable:**\n1. Start localGPT via Docker (`./start-docker.sh`) or Direct Development (`python run_system.py`).\n2. Navigate to your chat session settings in the frontend.\n3. Find and enable **\"Prune irrelevant sentences\"**. Alternatively, call the API with `provence_prune=true` if interacting directly. [Confidence: 50%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "No shipped profile has a provence block, so pruning is off by default.", "label_grounded": false, "qwen4b_votes": 1, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_on_acq_q04", "question": "What proportion of the target company's turnover comes from its single biggest client?", "answer": "Answer: Based on the provided documents, the single largest client is **MegaCorp**, which accounts for **28%** of StartupXYZ's revenue. This specific concentration risk (HIGH priority) was explicitly noted in both the Due Diligence Report and the Risk Assessment Memo regarding TechCorp Industries' proposed acquisition of StartupXYZ LLC as of December 20, 2024. The documents further specify that MegaCorp represents a significant portion of the company's business structure, contributing to an annual revenue at risk of $3.4M in the event their contract is affected by change-of-control clauses or other conditions. No information was found regarding specific turnover figures for any secondary clients beyond those requiring consent (DataFlow and CloudTech), nor details on MegaCorp's exact contribution relative to total turnover aside from this percentage. [Confidence: 90%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "The largest customer, MegaCorp, accounts for 28% of revenue.", "label_grounded": true, "qwen4b_votes": 0, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_on_acq_q07", "question": "How much did the parties pay to file under Hart-Scott-Rodino?", "answer": "Answer:\nAccording to the **Regulatory Approval Letter** from the Federal Trade Commission dated January 28, 2025, the HSR Filing Fee for TechCorp Industries' acquisition of StartupXYZ LLC was **$30,000**. The letter explicitly lists this amount under \"FILING DETAILS\" alongside a transaction value of $45,000,000. This fee corresponds to the Hart-Scott-Rodino Antitrust Improvements Act of 1976 filing required before closing could proceed on March 1, 2025.\n\n**Financial Adjustments Memo**: The **Financial Adjustments Memo**, issued by the Finance Department on December 23, 2024, and referenced in various documents, details specific price reductionsโ€”such as working capital shortfalls, undiscovered debt, and deferred revenueโ€”that collectively lower the original $45M purchase price to an adjusted total of approximately **$43.33 million** plus a mandatory escrow reserve for risks like customer concentration identified in the separate Risk Assessment Memo. It outlines these financial corrections relative to the Acquisition Agreement's initial terms while proposing revised payment structures including cash, stock, and earnout components contingent on performance metrics defined in Exhibit C.\n\n(c) Earnout: $5,000,000 (unchanged, per Exhibit C - Earnout Terms ) (d) Escrow: $1,300,000 (18-month release schedule) These adjustments have been discussed with Seller's representatives and are subject to final negotiation. Please refer to Document: Closing Checklist for timeline and requirements.\n\nContext: The current candidate is ThinknCheck from UPenn, which features a 1B parameter model designed for reasoning tasks such as arXiv QA or math problems using Chain-of-Thought methods (78.1 BAcc). However, this evaluation suggests that the proposed ThinknCheck(8B) variant requires approximately 16 GB of memory, exceeding significantly beyond the available budget constraints for this specific seam. Consequently, despite being an existing Apache-2.0 project with high performance on its target domain tasks, the larger model is deemed impractical due to resource limitations compared to smaller alternatives like TinyLlama.\n\n---\n\n| Candidate | Verdict |\n|-----------------------------------------------------|-----------|\n| ThinknCheck(arXiv 2604.01652, UPenn, 1B, 78.1 BAcc) | | \nContext: The provided chunk notes that an 8-billion parameter model exists under Apache-2.0 but is far too large for the current budget due to its ~16 GB memory requirement. This observation follows another failed candidate, a hate speech classifier from arXiv ID 2604.01652 at UPenn, which also lacks the required answer-vs-evidence entailment scoring capability. Both items represent unsuitable alternatives within this search for appropriate model candidates given resource constraints and task specificity.\n\n---\n\n| Exists, Apache-2.0 โ€” but 8B / ~16 GB, far over the budget this seam is for. |\n|-------------------------------------------------------------------------------| \nContext: The document explores machine learning models suitable for scoring answer-vs-evidence entailment within a constrained budget environment, noting that existing 38-million-parameter options like this specific RoBERTa variant are unsuitable due to their Apache-2.0 licensing conflicts or irrelevant training tasks focused on hate speech rather than logical entailment. While some larger alternatives exist in the multi-billion parameter range with appropriate licenses but exceed current resource limits, a more viable option identified is MIT's 369-million-parameter model which offers generic NLI capabilities without custom code and fits within the budgetary constraints.\n\n---\n\n| Exists, 38M, Apache-2.0 โ€” but it is ahate/abuse/profanityRoBERTa classifier. Wrong task: it does not score answer-vs-evidence entailment. |\n|---------------------------------------------------------------------------------------------------------------------------------------------| \n\nContext: This excerpt presents Morrison & Associates' due diligence report for TechCorp Industries regarding the proposed acquisition of StartupXYZ LLC as of December 20, 2024. The document provides a high-level executive summary covering six key areas: financial performance validated against adjustment memos; intellectual property rights confirmed via an IP certification letter from PatentWatch Legal Services and legal opinions on litigation; employee retention risks addressed through transition plans; contract compliance including change-of-control provisions; regulatory standing requiring HSR filing timelines; and final recommendations contingent upon specific conditions. It serves as the primary briefing document preceding detailed attachments like schedules, letters of verification, and risk assessments cited within its sections.\n\n---\n\nDUE DILIGENCE REPORT CONFIDENTIAL DUE DILIGENCE REPORT Prepared for: TechCorp Industries, Inc. Subject: StartupXYZ LLC Date: December 20, 2024 Prepared by: Morrison & Associates, LLP EXECUTIVE SUMMARY This report summarizes our findings from the due diligence investigation of StartupXYZ LLC in connection with the proposed acquisition described in the Document: Acquisition Agreement . 1. FINANCIAL REVIEW 1.1 Revenue for FY2024: $12.3 million (growth of 45% YoY) 1.2 EBITDA: $2.1 million (17% margin) 1.3 Cash position: $3.2 million as of November 30, 2024 1.4 Outstanding debt: $1.5 million (detailed in Exhibit A - Financial Terms of the Acquisition Agreement) KEY FINDING: Financial statements are materially accurate. Minor adjustments recommended as noted in Document: Financial Adjustments Memo . 2. INTELLECTUAL PROPERTY 2.1 StartupXYZ holds 12 patents related to AI/ML technology 2.2 All patents verified as valid per Document: IP Certification Letter 2.3 No pending litigation affecting IP (confirmed in Document: Legal Opinion Letter ) 2.4 Full IP inventory in Schedule 1 - IP Assets of the Acquisition Agreement 3. EMPLOYEE MATTERS 3.1 Total employees: 47 (32 engineering, 8 sales, 7 operations) 3.2 Key employee retention risk: HIGH for 5 senior engineers 3.3 Retention bonuses recommended per Schedule 3 - Employee Transition Plan 3.4 No pending employment disputes 4. MATERIAL CONTRACTS 4.1 23 active customer contracts reviewed (see Schedule 2 - Material Contracts ) 4.2 3 contracts contain change-of-control provisions requiring consent 4.3 Largest customer (MegaCorp) accounts for 28% of revenue - concentration risk noted in Document: Risk Assessment Memo 5. REGULATORY COMPLIANCE 5.1 Company is compliant with all applicable regulations 5.2 HSR filing required - timeline in Document: Regulatory Approval Letter 6. RECOMMENDATIONS Based on our findings, we recommend proceeding with the acquisition subject to: (a) Obtaining customer consents for change-of-control contracts (b) Implementing retention packages for key employees (c) Addressing items in Document: Financial Adjustments Memo Respectfully submitted, Morrison & Associates, LLP\n\nContext: This chunk outlines the \"Design Rationale\" framework used by localGPT since 2026-08-09, which structures technical decisions into three artifacts: Evidence (primary source research), Our own eval (internal metrics like nDCG@10), and A decision (actionable defaults). It emphasizes that literature claims are subordinate to internal testing results, where data-driven disagreements override published benchmarks without verification. The text sets the stage for specific component analyses by detailing how parsing/OCR choices were made after evaluating VLM spikes against local performance constraints.\n\n---\n\nDesign Rationale Revision: 2026-08-09. Roadmap item 3.1 . This document answers one question per component: why does localGPT do it this way? It describes only what ships in the current tree. Anything planned, proposed or merely promising lives in research_roadmap.md and improvement_plan.md , not here. The method Three artefacts, in a fixed order: Evidence โ€” research/ holds three August-2026 sweeps of primary sources (papers, model cards, first-party engineering blogs), each claim graded established / emerging / contested and each carrying its own \"could not verify\" appendix. These documents describe the field, not this repo . Our own eval โ€” eval/ holds a 72-query gold set over three corpora, a recall/nDCG runner, a binary groundedness judge validated against hand labels, and an end-to-end smoke test. See eval/README.md . A decision โ€” nothing changes a default without a measured delta on (2). The decisions and their numbers are in eval/DECISIONS.md and eval/decisions/ . The order matters because it repeatedly produced the opposite answer from the literature. Three times in one week: component-map-2026.md ยง5.1 , ยง6.5) | bge-reranker-v2-m3 on top of the new first stage: โˆ’0.022 nDCG@10 on mixed , โˆ’0.058 on docs , for ~1.6 s/query | Reranking off by default ( DECISIONS.md ยง1 ) | component-map-2026.md ยง6.2 ) | On the 6 queries that genuinely decompose: โˆ’0.046 ( max ) / โˆ’0.012 ( mean ) nDCG@10 | Shape change shipped; sub-query scoring enabled by no profile ( phase2-pipeline.md ยง3 ) | Qwen3-Embedding-4B microsoft/harrier-oss-v1-0.6b as the default ( embedder.md gate section ) | That is the whole method: the literature nominates candidates, our gold set decides. Where the two disagree, the gold set wins and the disagreement is written down rather than smoothed over. Every number below is traceable to a file under eval/ . Where a claim has no number, it says so. 1. Parsing / OCR What ships. \nContext: This excerpt details the \"What ships\" strategy for document parsing within localGPT's 2026 design revision, specifying that PyMuPDF acts as a text-layer probe to gate OCR usage while docling handles conversion without reliance on Vision-Language Models (VLMs). It justifies this orchestration-only approach by citing evidence against adopting GLM-OCR despite its perceived superiority in certain cases like scanned invoices due to technical defects and the lack of an actual OCR evaluation corpus. The text emphasizes that decisions are driven by empirical performance data rather than leaderboard positions or theoretical promises, strictly adhering to a \"gold set\" decision framework where unverified claims are rejected as GO-LATER items.\n\n---\n\nEvery document goes through docling ( rag_system/ingestion/document_converter.py ). PDFs take one of two converters: a text-layer probe ( _pdf_has_text , PyMuPDF) decides whether OCR runs at all โ€” if any page has extractable text, the whole document is converted without OCR. DOCX / HTML / MD go through a third general converter; .txt is read directly and fenced. Conversion returns (markdown, metadata, DoclingDocument) so the chunker can use the element tree rather than re-parsing markdown. The OCR engine is probed, not configured : build_ocr_options() walks OcrMac โ†’ EasyOCR โ†’ RapidOCR โ†’ tesserocr โ†’ tesseract-cli and picks the first whose backend is actually importable on this host. There is no VLM parser. Why. The 2026 evidence ( component-map-2026.md ยง1.5 ) says the local pipeline is \"Docling as orchestration + a 0.9โ€“1.2B specialist VLM as the parsing engine\", and that traditional parsers \"survive as the fast path, not the quality path\". We ship the orchestration half and not the VLM half, on purpose: The VLM spike ( glm-ocr-spike.md ) is GO-LATER , not NO-GO. It demonstrated a real, large win on the one document class that matters โ€” a degraded scanned invoice where GLM-OCR read 30/30 table cells and the current chain lost every price and 4 of 5 part numbers. Three defects block adoption: Ollama's Modelfile ignores prompts (so GLM-OCR's table/formula modes are unreachable), some pages are deterministically transcribed twice, and docling flattens the model's pipe tables to tables: 0 . None of them is code we should write blind. There is no OCR eval. eval/corpora/ contains only digital-born PDFs with clean text layers, so nothing in the harness exercises the OCR branch. Adopting a parser on leaderboard position alone would violate the gate this repo runs on โ€” and the spike showed exactly why: the roadmap's original \"#1 on OmniDocBench, beats GPT-5.2 by ~10 points\" line did not survive source verification (GLM-OCR is third on v1.6_full, behind PaddleOCR-VL-1.6 and MinerU2.5-Pro).\n\nContext: The provided chunk notes that an 8-billion parameter model exists under Apache-2.0 but is far too large for the current budget due to its ~16 GB memory requirement. This observation follows another failed candidate, a hate speech classifier from arXiv ID 2604.01652 at UPenn, which also lacks the required answer-vs-evidence entailment scoring capability. Both items represent unsuitable alternatives within this search for appropriate model candidates given resource constraints and task specificity.\n\n---\n\n| Exists, Apache-2.0 โ€” but 8B / ~16 GB, far over the budget this seam is for. |\n|-------------------------------------------------------------------------------| \nContext: The document explores machine learning models suitable for scoring answer-vs-evidence entailment within a constrained budget environment, noting that existing 38-million-parameter options like this specific RoBERTa variant are unsuitable due to their Apache-2.0 licensing conflicts or irrelevant training tasks focused on hate speech rather than logical entailment. While some larger alternatives exist in the multi-billion parameter range with appropriate licenses but exceed current resource limits, a more viable option identified is MIT's 369-million-parameter model which offers generic NLI capabilities without custom code and fits within the budgetary constraints.\n\n---\n\n| Exists, 38M, Apache-2.0 โ€” but it is ahate/abuse/profanityRoBERTa classifier. Wrong task: it does not score answer-vs-evidence entailment. |\n|---------------------------------------------------------------------------------------------------------------------------------------------| \nContext: The document discusses NLI models available for grounded claim verification tasks, contrasting different versions by their source size, task specificity, and presence of custom code. While a 38M Hate/Abuse classifier is noted as unsuitable due to its misaligned objective, the specific chunk highlights an MIT dataset at 369 MB designed for generic Natural Language Inference with no custom code modifications. This model serves as one alternative option alongside other benchmarks and purpose-built systems evaluated in the ThinknCheck context.\n\n---\n\n| โœ… MIT, 369 MB, no custom code. Generic NLI. |\n|-----------------------------------------------| \n\nContext: This section details the RAG system's data architecture components, specifically the LanceDB vector store structure for storing document chunks alongside their metadata and text indexes, as well as file storage for uploaded documents and logs. Immediately following this technical breakdown are configuration defaults in `rag_system/main.py` that define environment variables like `EMBEDDING_MODEL`, which points to a HuggingFace model such as `microsoft/harrier-oss-v1-0.6b`. The text further elaborates on these models, listing alternative embeddings and rerankers for optimization while noting critical constraints regarding vector width compatibility during indexing.\n\n---\n\n3.2 LanceDB ( ./lancedb , override with LANCEDB_PATH ) lancedb/\nโ”œโ”€โ”€ text_pages_v4 -- default table (storage.text_table_name)\nโ”œโ”€โ”€ text_pages_<index_id> -- one table per index created via POST /indexes\nโ””โ”€โ”€ <table>_lc -- late-chunk vectors for the table above Each table stores chunk_id , text , document_id , chunk_index , metadata (JSON) and vector , plus a native full-text index on text . 3.3 File system shared_uploads/ -- uploaded documents (<uuid>_<original name>)\nindex_store/overviews/<id>.jsonl -- per-index / per-session document overviews\nindex_store/overviews/overviews.jsonl -- global fallback overview file\nlogs/ -- run_system.py service logs + run_system.pid 4. Models 4.1 Configured defaults ( rag_system/main.py ) OLLAMA_CONFIG = {\n \"host\": os.getenv(\"OLLAMA_HOST\", \"http://localhost:11434\"),\n \"generation_model\": os.getenv(\"GENERATION_MODEL\", \"qwen3.5:9b\"),\n \"enrichment_model\": os.getenv(\"ENRICHMENT_MODEL\", \"qwen3.5:4b\"),\n}\n\nEXTERNAL_MODELS = {\n \"embedding_model\": os.getenv(\"EMBEDDING_MODEL\", \"microsoft/harrier-oss-v1-0.6b\"),\n \"reranker_model\": os.getenv(\"RERANKER_MODEL\", \"Qwen/Qwen3-Reranker-4B\"),\n} qwen3.5:9b (Ollama) | Final answers, sub-answer composition, direct answers | qwen3.5:4b (Ollama) | Agent triage (the only LLM router), query decomposition, contextual enrichment, document overviews, verification | microsoft/harrier-oss-v1-0.6b (HuggingFace, MIT, 1024 dims) | Index and query embeddings | Qwen/Qwen3-Reranker-4B (HuggingFace, own yes/no-logit scorer) | Reranking retrieved chunks โ€” off by default , loaded lazily only when switched on ( ../eval/DECISIONS.md ) | naver/provence-reranker-debertav3-v1 (HuggingFace) | Opt-in sentence-level pruning | Approximate footprints published by the model authors (not measured here): \nContext: This section details specific model selection configurations within the RAG system, covering approximate memory footprints for various LLMs (like qwen3.5:9b) alongside alternative embedding and reranking models such as harrier-oss or BGE-Reranker. It emphasizes critical operational constraints regarding runtime flexibility, specifically warning that changing embedding models necessitates full re-indexing to prevent vector mismatch errors in LanceDB tables. Finally, it explains the hierarchical model selection logic at both request-level (per session) and index-level levels defined by environment variables and server metadata.\n\n---\n\nqwen3.5:9b โ‰ˆ 6.6 GB at Q4, qwen3.5:4b โ‰ˆ 3.4 GB, qwen3.6:27b โ‰ˆ 17 GB, microsoft/harrier-oss-v1-0.6b โ‰ˆ 1.2 GB, Qwen/Qwen3-Embedding-4B โ‰ˆ 8 GB in bf16, Qwen/Qwen3-Embedding-0.6B โ‰ˆ 1.2 GB, Qwen/Qwen3-Reranker-4B โ‰ˆ 7.5 GB. 4.2 Documented alternatives qwen3.6:27b (high-end), qwen3.5:4b (light) | qwen3.5:2b (light) | Qwen/Qwen3-Embedding-4B (2560 dims, 32K context โ€” for multilingual / long-context corpora), Qwen/Qwen3-Embedding-0.6B (1024 dims, light) | BAAI/bge-reranker-v2-m3 (cross-encoder, low latency โ€” only pays off with a weaker embedder than the default), answerdotai/answerai-colbert-small-v1 (late interaction โ€” also set reranker.model_type: \"colbert\" ), Qwen/Qwen3-Reranker-0.6B Set them with the GENERATION_MODEL / ENRICHMENT_MODEL / EMBEDDING_MODEL / RERANKER_MODEL environment variables, or edit rag_system/main.py . โš ๏ธ Changing the embedding model requires re-indexing. Vector width is derived from the loaded model, and VectorIndexer raises rather than appending mismatched vectors to an existing LanceDB table. Width alone is not a sufficient check โ€” harrier-oss-v1-0.6b and Qwen3-Embedding-0.6B are both 1024 dims โ€” so every table also records the embedding model that wrote it, and indexing into or querying it with a different one raises EmbedderMismatchError . Ollama embedding tags are also supported: select_embedder() treats a name containing / as a HuggingFace repo and anything else as an Ollama tag. 4.3 Model selection at runtime Per request โ€” model on POST :8000/sessions/{id}/messages \nContext: The provided chunk details dynamic model management across different system paths while outlining current limitations for multimodal features: it explains how request-level generation models override defaults (with backend compatibility constraints), describes per-index embedding switches, and clarifies that vision capabilities are absent but can be extensible; this technical overview is followed by a structured definition of the available pipeline configurations, specifically comparing the \"default\" profile's hybrid search setup against other profiles.\n\n---\n\nand on the RAG API chat endpoints overrides the generation model for that request only. The RAG API rejects ids that do not match the active backend (an Ollama tag will not be forced onto a WatsonX deployment). Per index โ€” when an index records an embedding_model in its metadata, the RAG API switches the retrieval pipeline's embedder to it before querying that index. Generation model precedence on the gateway's direct-LLM path: request model โ†’ the session's model_used โ†’ GENERATION_MODEL . 4.4 Vision / multimodal โ€” not integrated There is no vision model in the configuration and no multimodal path in the pipelines: PDF parsing and OCR are handled entirely by Docling. Models such as GLM-OCR or Qwen3-VL could be added as an extension; wiring them up is not done today. 4.5 Alternative LLM backend: WatsonX LLM_BACKEND=watsonx swaps the Ollama client for WatsonXClient ( WATSONX_CONFIG : WATSONX_API_KEY , WATSONX_PROJECT_ID , WATSONX_URL , WATSONX_GENERATION_MODEL , WATSONX_ENRICHMENT_MODEL ). It requires pip install ibm-watsonx-ai โ€” the root requirements.txt lists it as an optional, commented dependency. Embedding and reranking still run locally through HuggingFace. See ../WATSONX_README.md . 5. Pipeline Configurations PIPELINE_CONFIGS in rag_system/main.py contains exactly two profiles, default and fast . RAG_CONFIG_MODE selects the one the RAG API server uses (default default ); an unknown value silently falls back to default . factory.get_pipeline_config() hands out a deep copy, so runtime overrides never mutate the master config. 5.1 default\n\nContext: This chunk summarizes the technical implementation and empirical rationale for defaulting reranking off while offering an opt-in AI reranker feature that uses a Qwen3-based cross-encoder instead of relying on the library's standard backend. It argues against automatically enabling reranking because, with improved first-stage retrieval quality, even high-performing models like `bge-reranker-v2-m3` can degrade results if applied universally without tuning. The decision reflects a balance between the strong theoretical ROI of reranking and its substantial latency cost for single-user environments, noting that significant improvements would require specific triggers such as embedder changes or future library updates rather than constant global application.\n\n---\n\nmisses to calibrate against. The firing rate drifted 9.7% โ†’ 11.1% as the corpus grew, so it is a property of the corpus, not a constant โ€” re-check it after any embedder change. 6. Reranking posture What ships. The default profile ships reranker.enabled = False ( rag_system/main.py::PIPELINE_CONFIGS ), and the UI \"AI reranker\" toggle defaults off to match ( src/components/ui/session-chat.tsx ). When the toggle is switched on, it lazily loads Qwen/Qwen3-Reranker-4B through the in-repo QwenRerankerScorer ( rag_system/rerankers/reranker.py ), routed either by explicit reranker.model_type: \"qwen3\" or by model name ( retrieval_pipeline.py::_get_ai_reranker ). Any other model still goes through the rerankers library. A reranker that fails to load logs a warning and is skipped โ€” there is no fallback reranker. Why this is the most counter-intuitive decision in the repo. The evidence is about as strong as evidence gets: reranking is \"the single highest-ROI component in the stack\", +17.2 pp MRR@3 on T2-RAGBench, โˆ’1.7 EM when removed from a local 7B ablation, and the 2026 recommendation is a flat \" YES, unconditionally \" ( component-map-2026.md ยง5.1, ยง6.5 ). Our gold set says otherwise, for a specific and explicable reason. The joint matrix ( DECISIONS.md ยง1 ), mixed corpus, same 20 first-stage candidates reordered by each: BAAI/bge-reranker-v2-m3 Qwen/Qwen3-Reranker-4B Two findings, both load-bearing: The cheap cross-encoder now hurts. bge-reranker-v2-m3's famous +0.232 on docs was largely a repair job on a weak first stage . Improve the first stage and the repair becomes damage. This is exactly why 1.1 and 1.2 could not be decided independently and were re-measured jointly in one re-index window. The good reranker is a real win and still too slow to default on. +0.062 nDCG@10 on mixed , +0.173 on docs \nContext: This section evaluates reranker options for the retrieval pipeline, contrasting a high-performance but slow model (Qwen3-Reranker-4B) against cheaper alternatives rejected due to negligible gains or latency costs. It details technical implementation challenges where loading models incorrectly results in untrained noise and explains specific configuration requirements to avoid this failure mode. The discussion concludes by identifying conditionsโ€”such as embedder changes, latency tuning, future library releasesโ€”that would necessitate re-evaluating these decisions.\n\n---\n\nโ€” the largest single quality win in this repo's eval history โ€” for ~12.7 s per query and 7.5 GB of resident weights on a single-user, single-threaded server. That is worth paying on demand , not on every message. Qwen/Qwen3-Reranker-0.6B was rejected outright: +0.021 nDCG@10 over bge (one to two queries out of 72, inside the noise band) bought with 1.5โ€“2.8ร— the latency, while losing recall@10 on both corpora ( reranker.md ยง7 ). One integration finding worth keeping. rerankers 0.10.0 has no Qwen3-Reranker backend. Loading one through the shipped cross-encoder path builds a Qwen3ForSequenceClassification with a randomly initialised score head โ€” had the batching not thrown, it would have returned untrained noise while printing \"AI reranker initialized successfully\". QwenRerankerScorer implements the model card's actual scheme (causal LM, left padding, chat template, softmax over the yes / no logits at the final position), and the name-based route in the loader exists specifically so no configuration can reach the random head ( reranker.md ยง1 ). What would change the decision. Three concrete triggers: Re-run the A/B if the embedder changes. The reranker's headroom is largest exactly where the first stage is weakest, so this decision is a function of ยง3 and expires with it. Tune the latency knobs. Batch size (8) and the 2048-token truncation cap are both untuned, and reranking only the top 10 candidates instead of 20 is unmeasured. Either could move the 12.7 s materially โ€” that is unmeasured work, not a promise. A later rerankers release may add a Qwen3 backend, at which point the name-based route should be revisited. Note also what no reranker fixed: docs_d09 and docs_d17 degrade under every model tested. They are a query-understanding problem, and this section is not where they get solved. 7. Query decomposition What ships. The first stage always runs once, on the full original query. Sub-queries are used at the rerank stage, where each candidate is scored against every sub-query and the scores are aggregated by query_decomposition.rerank_aggregate ( \nContext: This section details query decomposition strategies within a retrieval pipeline where sub-query scoring at rerank is currently disabled because no active profiles enable it; this restriction stems from evidence showing semantic dilution when decomposing queries early, resulting in degraded performance for multi-hop cases despite overall corpus gains coming from single-sub-query rewriting. While the current implementation favors strict work reduction over experimental features like conditional decomposition, a future re-evaluation of query-decomposition benefits could alter these defaults if sufficient data proves its necessity again alongside routing logic layers that separate deterministic gateway filtering from LLM-based agent triage decisions.\n\n---\n\n\"mean\" default, \"max\" available) โ€” retrieval_pipeline.py::_rerank_stage . First-stage fan-out survives only behind the pre-existing compose_from_sub_answers flag, which the default profile sets to true because that path needs a separate answer per sub-question, which one shared candidate set cannot produce ( rag_system/agent/loop.py ). Consequence, stated plainly: no shipped profile enables sub-query scoring at rerank. With compose_from_sub_answers: true the aggregation path is never reached, and with reranking off there is no rerank stage at all. Why. The evidence puts decomposition at \"conditional โ€” multi-hop, applied at rerank \", noting that decomposition at initial retrieval dilutes the query semantically ( component-map-2026.md ยง6.2, ยง6.5 ). We shipped the shape change and measured the payload negative. On docs with Qwen3-Reranker-4B , only 6 of 24 queries decompose into more than one sub-query. On exactly those 6, scoring against sub-queries at rerank is worse under both aggregates: 0.8862 โ†’ 0.8406 ( max , โˆ’0.046) and 0.8740 ( mean , โˆ’0.012) . The whole-corpus \"gain\" comes entirely from the 18 single-sub-query rows, where the win is query rewriting , not decomposition ( phase2-pipeline.md ยง3 ). The shape change ships anyway because it is a strict reduction in work โ€” the aggregate path used to issue N first-stage retrievals and now issues one โ€” and because the structural check passed: the first-stage number is byte-identical across all three arms, proving decomposition can no longer touch it. What would change the decision. n_effective = 6 queries on one corpus . That is too small to call the 2026 MultiConIR/SSRB finding wrong; it is big enough to say it did not reproduce here, which is why nothing was switched on. mean stays the default aggregate on the \"less bad\" argument, not a positive one. 8. Routing / triage What ships โ€” two layers, exactly one of which calls an LLM. Layer 1, the gateway ( backend/server.py::should_use_rag ) is deterministic and makes no network call:\n\nContext: The provided chunk details the implementation of a two-layer system for routing queries between direct LLM responses and RAG processes, comprising an initial gateway filter for smalltalk and metadata followed by an agent-level decision based on document overviews. This mechanism is supported in contrast to pre-retrieval LLM routing strategies, which research identifies as inefficient and superior to complex ML approaches like discriminative classifiers are preferred at low cost due to their simplicity. Ultimately, the design intentionally biases toward sending queries to RAG because false negatives incur more significant costs than false positives when a fallback answer is required anyway.\n\n---\n\nforce_rag โ†’ RAG; no linked indexes โ†’ direct LLM; whole-message smalltalk (โ‰ค 6 words, anchored allowlist, must contain a core phrase) or assistant-meta (\"who are you\", \"what model are you\") โ†’ direct LLM; everything else โ†’ RAG. Unit-tested at 155/155 in backend/test_gateway_routing.py . Layer 2, the agent ( rag_system/agent/loop.py::_triage_query_async ) is the system's single LLM routing layer: document overviews + the utility model decide rag_query vs direct_answer , with \"history exists โ†’ rag_query \" as a shortcut and an LLM fallback when no overviews are loaded. _normalize_triage collapses anything that is not an explicit direct_answer to rag_query , so a small model still emitting the retired graph_query label lands on the RAG path. Why. Pre-retrieval LLM routing is the weakest measured pattern of 2026, and three independent sources agree: four ML approaches to pre-retrieval routing all failed because \"the need for augmentation cannot be determined from the query alone\"; rule-based retriever routing lost to fixed hybrid by 1.8 EM; and TF-IDF+SVM matches or beats neural and LLM routers at ~zero cost ( component-map-2026.md ยง7.1, ยง7.3, ยง7.4 ). The four-year pattern in ยง7.3 is \"a small discriminative classifier is the right tool; an LLM router is rarely justified.\" The bias toward over-sending to RAG is deliberate and cheap: agent triage runs on every forwarded request and can still answer directly, so a false \"use RAG\" costs one call on a model that would have been called anyway, while a false \"answer directly\" costs an unanswerable question. The gateway is a smalltalk filter in front of the decision-maker, not the decision-maker. Measured ( phase2-gateway.md ยง3 ): mean routing decision 750.613 ms โ†’ 0.002 ms over the same 20 messages, with 20/20 decision agreement with the LLM router it replaced. The deleted keyword fallback was worse than the old docs claimed โ€” it matched greetings by substring , so 'hi' matched w hi ch , t hi s and mac hi ne , routing 7 of 8 real Atlas-7 questions to the direct LLM. That is why it was deleted \nContext: The specific chunk details verification as an enabled feature that relies on either LLM prompts or local NL models (like MiniCheck) instead of previously planned but unavailable candidates like ThinknCheck to ground answers against retrieved evidence. It contrasts this practical implementation with past routing failures, such as the deleted test/check keyword fallback and unverified graph_query labels which incorrectly routed queries directly to an LLM rather than RAG. While a default verifier remains unchanged from prior versions, current lightweight substitutes were tested on hand-labeled cases without becoming new defaults due to mixed validation results or availability issues in other routing contexts.\n\n---\n\nrather than patched, and why \"messages containing test/check route RAG\" is now a regression test. What would change the decision. improvement_plan.md ยง2.1 (embed and cache document overviews for a cosine pre-check) and ยง2.2 (session-level routing memo) both still stand for the agent layer, which is now the only per-query LLM routing call. The evidence's own carve-out is that routing still pays for pipeline depth selection, implemented as a post-retrieval cascade โ€” which is what ยง5's retry is. 9. Verification What ships. Verification is on in default ( verification: {enabled: True} ). The shipped backend is an LLM prompt on the utility model ( rag_system/agent/verifier.py::Verifier.verify_async ), returning a JSON verdict and a confidence score; a low-confidence or ungrounded verdict appends a warning to the answer ( agent/loop.py ). A seam exists for a local model: VERIFIER_MODEL / verification.model swaps in LocalNLIVerifier , which sentence-splits the answer, scores each sentence against the retrieved evidence as premise, and takes the minimum โ€” one unsupported sentence makes the answer ungrounded, matching the binary semantics eval/judge.py already uses. A model that cannot be loaded raises , printing the availability table, rather than falling back: a verifier that silently is not the verifier you configured is worse than an error. The default is unchanged. Why. Verification helps as an external check, and the 2026 result is that a 4-bit 1B verifier (ThinknCheck, 78.1 BAcc) now beats the 7B 2024 SOTA, which would make per-answer grounding cheap enough to always run ( component-map-2026.md ยง9.2 ). Both of the roadmap's named candidates failed availability checks against the HuggingFace Hub API: ThinknCheck has no public weights (zero models returned; the paper links no release), and Granite Guardian is either 8B/~16 GB or โ€” in its 38M form โ€” a hate/abuse classifier, the wrong task entirely. Two substitutes were wired and exercised rather than left as a stub, and both were run against all 20 hand-labelled cases in eval/judge_validation.jsonl ( phase2-pipeline.md ยง4 ): lytang/MiniCheck-DeBERTa-v3-Large [Confidence: 95%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "The HSR filing fee was $30,000.", "label_grounded": true, "qwen4b_votes": 5, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_on_acq_q09", "question": "How does the borrowing the reviewers first disclosed compare with the extra borrowing found later, and what was the later amount?", "answer": "Answer:\nBased on the **Financial Adjustments Memo** issued by the Finance Department on December 23, 2024 (relative to a StartupXYZ acquisition), there is no information regarding \"borrowing,\" loan repayments, or interest expenses. The document discusses adjustments related to debt found during due diligence, but these are classified as liabilities rather than borrowing costs in the context of the question asked.\n\nHowever, regarding **liabilities** that reduce the purchase price (which may be what you are referring to if \"borrowing\" was a misstatement for \"debt\"):\n* Previously disclosed debt: $1,500,000\n* Additional identified debt found later (capital lease obligations): $175,000\n\nThe comparison is that the additional debt ($175,000) represents an **uncovered** portion of liabilities not seen in initial disclosures. The total liability adjustment resulted from these two amounts combined:\n* Total Debt Adjustment: (**$175,000**) to reduce the purchase price.\n\nIf you meant **\"borrowing\"** as \"working capital adjustments\" (cash flow needs):\n* Target working capital: $1,200,000 \n* Estimated closing working capital: **$980,000** \n* Adjustment (shortfall/needs adjustment): (**$220,000**)\n\nThere is no mention of \"borrowing the reviewers first disclosed\" in any form within these documents. Please clarify if you meant **\"debt identified\"** or another term to avoid further confusion. [Confidence: 90%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "Due diligence disclosed $1.5 million of debt; a further $175,000 of capital lease obligations was identified afterwards.", "label_grounded": true, "qwen4b_votes": 0, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_on_acq_q12", "question": "How large is the target's workforce and how is it split between functions?", "answer": "Answer:\nI could not find that information in the provided documents.\n\n***General knowledge**: The retrieved snippets focus entirely on technical documentation for a Retrieval-Augmented Generation (RAG) system, detailing its architecture (LanceDB, SQLite), operational workflow (chunking, embedding, retrieval, reranking), configuration parameters, and specific LLM prompts used for tasks like document summarization and query decomposition. There is no mention of the target company's workforce size or functional splits.* [Confidence: 100%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "47 employees: 32 in engineering, 8 in sales, 7 in operations.", "label_grounded": false, "qwen4b_votes": 0, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_on_docs_d03", "question": "How many overviews does the overview router use?", "answer": "Answer: The system uses the **first 40 loaded overviews** for routing decisions. This occurs during the `Overview routing` step within the agent's triage process (`_triage_query_async`). When this stage is active, it constructs a \"DOCUMENT OVERVIEWS:\" block from these top 40 entries to interpolate into the router prompt before calling the utility model. If no per-index overview file exists for a session, the system falls back to reading `index_store/overviews/overviews.jsonl`. Conversely, if an index has been recently built without overviews enabled (or failed), or specifically in scenarios where \"No overviews on disk\" are present relative to the agent layer, the router returns immediately with no specific count applied beyond skipping that step. It is important to note that only this **agent-side** triage router reads these files; the deterministic backend gateway gate does not read overview files at all.\n\nContext:\n- The text explicitly states in `loop.py` line 612โ€“613 (referenced as `_triage_query_async`) that it builds a block from \"the first **40** loaded overviews\".\n > \"...otherwise it builds a DOCUMENT OVERVIEWS: block from the first **40** loaded overviews (`loop.py`:612-613)...\"\n\nThis detail is confirmed in multiple retrieval snippets describing the agent-side triage logic where \"overviews\" are aggregated into context for the LLM classification. [Confidence: 95%]", "gold_answer": "The overview router uses the first 40 loaded overviews.", "label_grounded": true, "qwen4b_votes": 5, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_on_docs_d05", "question": "How many characters is the overview input truncated to?", "answer": "Answer: The documentation states that when generating document overviews, the input is **truncated to 5000 characters**. This truncation occurs before the text is sent to the model for summarization.\n\nThis limit is defined in `overview_builder.py` (specifically at line :37) within the logic used by the overview builder during indexing. The system sends only the first N chunks of a documentโ€”defaulting to 5โ€”to generate this truncated, one-paragraph summary which is then stored and utilized by triage routers. [Confidence: 100%]", "gold_answer": "Overview input is the first N chunks truncated to 5000 characters.", "label_grounded": true, "qwen4b_votes": 2, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell1_on_docs_d13", "question": "how do I enable pruning in localGPT?", "answer": "Answer: Based on the provided documentation, there is **no direct configuration option or command** to explicitly \"enable\" pruning as a toggleable feature (such as `pruning.enabled` or similar) that users can easily switch on and off. However, the system implements an optimization technique called **\"provence sentence pruning\"** which reduces context by removing sentences with low relevance scores after retrieval but before answer synthesis.\n\nTo enable this specific form of pruning:\n* You must use the **production-ready \"default\" pipeline profile**. The configuration for `PIPELINE_CONFIGS[\"default\"]` includes `\"pruning\": {\"enabled\": True}` (implied as active in the standard setup).\n* This feature is located within the RAG API's retrieval and synthesis flow, specifically after **Late-chunk retrieval** and before answer synthesis (`SY`).\n\nIf you are looking for a way to control whether this happens, it appears to be part of the default \"default\" profile which includes pruning. There is no specific flag provided in `rag_system/main.py` profiles (like `\"fast\"` or `\"default\"`) described as simply toggling \"pruning on/off\" distinct from other features like verification or reranking. The documentation notes that keys such as `enable_enrich: false` can disable enrichment to speed up indexing, but for pruning specifically in the query path, it is enabled by default within the agent's synthesis step (`SY`) when using the standard configuration modes available out of the box (specifically looking at the `\"default\"` profile). [Confidence: 95%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "No shipped profile has a provence block, so pruning is off by default.", "label_grounded": false, "qwen4b_votes": 0, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell2_off_acq_q04", "question": "What proportion of the target company's turnover comes from its single biggest client?", "answer": "Answer:\nAccording to the provided documents, **MegaCorp** is identified as StartupXYZ LLC's largest customer. This single client accounts for **28% of revenue**.\n\nThis information is explicitly stated in the *Due Diligence Report* under Section 4 (Material Contracts), where it notes that \"Largest customer (MegaCorp) accounts for 28% of revenue - concentration risk noted.\" The *Risk Assessment Memo* further reinforces this figure, listing MegaCorp as representing **28%** of StartupXYZ's revenue and identifying the associated change-of-control clause in their contract as a high-priority risk.\n\nAdditional context from the documents clarifies that there are 3 total customer contracts requiring consent for changes of control (MegaCorp Inc., DataFlow Systems, and CloudTech Partners). While MegaCorp has an annual contract value of $3.4 million, which addresses the primary revenue concentration concern, specific data indicating whether other individual customers exceed this threshold or exactly how their revenues compare to one another is not provided in the retrieved snippets; therefore, based solely on the available information, MegaCorp remains cited as the singular largest customer with a 28% share of turnover. [Confidence: 100%]", "gold_answer": "The largest customer, MegaCorp, accounts for 28% of revenue.", "label_grounded": true, "qwen4b_votes": 4, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell2_off_acq_q12", "question": "How large is the target's workforce and how is it split between functions?", "answer": "StartupXYZ LLC has a total workforce of 47 employees. This team is distributed across three functions: Engineering with 32 employees, Sales with 8 employees, and Operations with 7 employees. The engineering department represents approximately two-thirds of the staff and includes five senior engineers considered critical for product development. [Confidence: 100%]", "gold_answer": "47 employees: 32 in engineering, 8 in sales, 7 in operations.", "label_grounded": true, "qwen4b_votes": 4, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell2_on_acq_q04", "question": "What proportion of the target company's turnover comes from its single biggest client?", "answer": "Answer:\nBased on the provided documents, **28%** of StartupXYZ LLC's revenue (turnover) comes from its largest customer.\n\nThis figure is explicitly identified in two separate reports as a high-priority risk due to concentration issues:\n* In the **Due Diligence Report** (Section 4.3), it states that \"Largest customer (MegaCorp) accounts for **28%** of revenue.\" The report specifically flags this under \"concentration risk noted in Document: Risk Assessment Memo.\"\n* This is corroborated by the **Risk Assessment Memo**, which details under Section 1.1 (\"Customer Concentration\") that \"**MegaCorp** represents **28%** of StartupXYZ revenue.\"\n\nThe documents further explain that MegaCorp holds this position within a portfolio where there are 23 active customer contracts reviewed in total, and specifically highlights the financial impact of losing such a client, noting an annual risk exposure of $3.4 million at closing due to their contract value. [Confidence: 95%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "The largest customer, MegaCorp, accounts for 28% of revenue.", "label_grounded": true, "qwen4b_votes": 0, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} +{"id": "cell2_on_acq_q12", "question": "How large is the target's workforce and how is it split between functions?", "answer": "Based on the provided documents from StartupXYZ LLC's Due Diligence Report, the company maintains a total workforce of 47 employees. This group is functionally split into Engineering with 32 staff members, Sales with 8, and Operations with 7. The report also notes that among these totals are five senior engineers critical to product development who face specific retention risks. No conflicting headcount information was found in other related memos. [Confidence: 90%] [Warning: Low confidence. Groundedness: False]", "gold_answer": "47 employees: 32 in engineering, 8 in sales, 7 in operations.", "label_grounded": true, "qwen4b_votes": 2, "source": "phase4-escalation-rerun 2026-08-12, manual adjudication ยง6"} diff --git a/eval/judge_validation.jsonl b/eval/judge_validation.jsonl new file mode 100644 index 00000000..871d9a40 --- /dev/null +++ b/eval/judge_validation.jsonl @@ -0,0 +1,20 @@ +{"id": "g01_brew_pressure", "label_grounded": true, "corpus": "atlas7", "failure_mode": null, "question": "What pressure does the brew boiler run at during extraction?", "answer": "The brew boiler operates at 9.2 bar during extraction.", "evidence": ["1. OPERATING SPECIFICATIONS\nThe brew boiler operates at a pressure of 9.2 bar during extraction. The steam boiler is maintained at 1.45 bar. The PID controller keeps brew water at 93.5 degrees Celsius with a tolerance of 0.4 degrees. The vibratory pump is rated for 52 watts continuous duty."]} +{"id": "g02_descaling", "label_grounded": true, "corpus": "atlas7", "failure_mode": null, "question": "How often does the machine need descaling?", "answer": "Descaling must be performed every 60 days when water hardness exceeds 120 ppm.", "evidence": ["2. MAINTENANCE SCHEDULE\nDescaling must be performed every 60 days when water hardness exceeds 120 ppm. The group head gasket (part MG-311) should be replaced every 14 months. Backflushing with Cafiza detergent is recommended weekly."]} +{"id": "g03_gasket", "label_grounded": true, "corpus": "atlas7", "failure_mode": null, "question": "Which part is the group head gasket and how often is it replaced?", "answer": "The group head gasket is part MG-311 and should be replaced every 14 months.", "evidence": ["2. MAINTENANCE SCHEDULE\nDescaling must be performed every 60 days when water hardness exceeds 120 ppm. The group head gasket (part MG-311) should be replaced every 14 months. Backflushing with Cafiza detergent is recommended weekly."]} +{"id": "g04_e42_procedure", "label_grounded": true, "corpus": "atlas7", "failure_mode": null, "question": "What does error E42 mean and how do I clear it?", "answer": "E42 indicates pump cavitation. Prime the pump by running 200 ml of water through the hot water wand, then power cycle the unit.", "evidence": ["3. ERROR CODES\nE11: Brew boiler thermistor open circuit. Replace sensor part TS-71.\nE23: Steam boiler overpressure. Check the OPV calibration at 12 bar.\nE42: Pump cavitation detected. Prime the pump by running 200 ml of water through the hot water wand, then power cycle the unit.\nE57: Flow meter reports zero pulses. Clean the inlet mesh filter."]} +{"id": "g05_warranty_length", "label_grounded": true, "corpus": "atlas7", "failure_mode": null, "question": "How long is the parts warranty, and what invalidates it?", "answer": "The Atlas-7 has a 36-month parts warranty, which is void if third-party descaling agents with more than 8 percent citric acid are used.", "evidence": ["4. WARRANTY\nThe Atlas-7 carries a 36-month parts warranty. The warranty is void if third-party descaling agents containing citric acid above 8 percent concentration are used. Warranty claims require the serial number, which is engraved under the drip tray on the left rail."]} +{"id": "g06_annual_leave", "label_grounded": true, "corpus": "hr", "failure_mode": null, "question": "How much annual leave does an employee below Grade 7 get?", "answer": "23 days of paid annual leave per calendar year.", "evidence": ["1. ANNUAL LEAVE\nEmployees below Grade 7 accrue 23 days of paid annual leave per calendar year. Employees at Grade 7 and above accrue 28 days. Accrual begins on the first day of employment and is credited monthly in arrears.\nA maximum of 5 unused annual leave days may be carried into the following year. Carried days expire on 31 March and are not paid out on expiry."]} +{"id": "g07_sick_pay", "label_grounded": true, "corpus": "hr", "failure_mode": null, "question": "How is sick leave paid?", "answer": "Sick leave is paid at 100 percent of base salary for the first 12 weeks of a single absence, then at 60 percent for a further 8 weeks.", "evidence": ["3. SICK LEAVE\nSick leave is paid at 100 percent of base salary for the first 12 weeks of a single absence, and at 60 percent for a further 8 weeks. A medical certificate is required once an absence exceeds 4 consecutive working days.\n4. PARENTAL LEAVE\nParental leave is 18 weeks per child, of which 6 weeks are fully paid. It must be taken before the child's third birthday. Parental leave may be split into no more than 3 separate blocks."]} +{"id": "g08_sabbatical", "label_grounded": true, "corpus": "hr", "failure_mode": null, "question": "What are the rules for taking an unpaid sabbatical?", "answer": "You need at least 4 years of continuous service. The sabbatical can be up to 90 days, requires 60 days written notice, and is approved by the Head of People Operations.", "evidence": ["7. UNPAID SABBATICAL\nEmployees with at least 4 years of continuous service may apply for an unpaid sabbatical of up to 90 days. Applications require 60 days written notice and are approved by the Head of People Operations. A sabbatical does not interrupt continuous-service accrual."]} +{"id": "g09_public_holiday", "label_grounded": true, "corpus": "hr", "failure_mode": null, "question": "What happens if I am rostered to work on a public holiday?", "answer": "You are paid at 1.5 times the normal rate and receive a substitute day off within the same quarter.", "evidence": ["8. PUBLIC HOLIDAYS\nNorthwind Robotics recognises 9 public holidays. An employee rostered to work on a public holiday is paid at 1.5 times the normal rate and receives a substitute day off within the same quarter."]} +{"id": "g10_contractors", "label_grounded": true, "corpus": "hr", "failure_mode": null, "question": "Does this leave policy cover agency contractors?", "answer": "No. Contractors engaged through an agency are not covered by this policy.", "evidence": ["9. EXCLUSIONS AND FORFEITURE\nAnnual leave is not paid out on resignation unless the employee has served more than 6 months. Leave taken without portal approval is recorded as unauthorised absence and is unpaid. Contractors engaged through an agency are not covered by this policy."]} +{"id": "u01_brew_pressure_wrong", "label_grounded": false, "corpus": "atlas7", "failure_mode": "wrong number", "question": "What pressure does the brew boiler run at during extraction?", "answer": "The brew boiler operates at 9.5 bar during extraction.", "evidence": ["1. OPERATING SPECIFICATIONS\nThe brew boiler operates at a pressure of 9.2 bar during extraction. The steam boiler is maintained at 1.45 bar. The PID controller keeps brew water at 93.5 degrees Celsius with a tolerance of 0.4 degrees. The vibratory pump is rated for 52 watts continuous duty."]} +{"id": "u02_descale_interval_wrong", "label_grounded": false, "corpus": "atlas7", "failure_mode": "wrong number", "question": "How often does the machine need descaling?", "answer": "Descaling must be performed every 90 days when water hardness exceeds 120 ppm.", "evidence": ["2. MAINTENANCE SCHEDULE\nDescaling must be performed every 60 days when water hardness exceeds 120 ppm. The group head gasket (part MG-311) should be replaced every 14 months. Backflushing with Cafiza detergent is recommended weekly."]} +{"id": "u03_boilers_swapped", "label_grounded": false, "corpus": "atlas7", "failure_mode": "swapped entity", "question": "What pressures do the two boilers run at?", "answer": "The steam boiler operates at 9.2 bar and the brew boiler is maintained at 1.45 bar.", "evidence": ["1. OPERATING SPECIFICATIONS\nThe brew boiler operates at a pressure of 9.2 bar during extraction. The steam boiler is maintained at 1.45 bar. The PID controller keeps brew water at 93.5 degrees Celsius with a tolerance of 0.4 degrees. The vibratory pump is rated for 52 watts continuous duty."]} +{"id": "u04_e42_extra_step", "label_grounded": false, "corpus": "atlas7", "failure_mode": "unsupported addition", "question": "What does error E42 mean and how do I clear it?", "answer": "E42 indicates pump cavitation. Prime the pump by running 200 ml of water through the hot water wand, then power cycle the unit, and run a full descaling cycle before the next extraction.", "evidence": ["3. ERROR CODES\nE11: Brew boiler thermistor open circuit. Replace sensor part TS-71.\nE23: Steam boiler overpressure. Check the OPV calibration at 12 bar.\nE42: Pump cavitation detected. Prime the pump by running 200 ml of water through the hot water wand, then power cycle the unit.\nE57: Flow meter reports zero pulses. Clean the inlet mesh filter."]} +{"id": "u05_sensor_part_transposed", "label_grounded": false, "corpus": "atlas7", "failure_mode": "wrong identifier", "question": "Which sensor do I replace for error E11?", "answer": "Replace sensor part TS-17.", "evidence": ["3. ERROR CODES\nE11: Brew boiler thermistor open circuit. Replace sensor part TS-71.\nE23: Steam boiler overpressure. Check the OPV calibration at 12 bar.\nE42: Pump cavitation detected. Prime the pump by running 200 ml of water through the hot water wand, then power cycle the unit.\nE57: Flow meter reports zero pulses. Clean the inlet mesh filter."]} +{"id": "u06_warranty_length_wrong", "label_grounded": false, "corpus": "atlas7", "failure_mode": "wrong number", "question": "How long is the parts warranty?", "answer": "The Atlas-7 carries a 24-month parts warranty.", "evidence": ["4. WARRANTY\nThe Atlas-7 carries a 36-month parts warranty. The warranty is void if third-party descaling agents containing citric acid above 8 percent concentration are used. Warranty claims require the serial number, which is engraved under the drip tray on the left rail."]} +{"id": "u07_annual_leave_wrong", "label_grounded": false, "corpus": "hr", "failure_mode": "wrong number", "question": "How much annual leave does an employee below Grade 7 get?", "answer": "Employees below Grade 7 accrue 25 days of paid annual leave per calendar year.", "evidence": ["1. ANNUAL LEAVE\nEmployees below Grade 7 accrue 23 days of paid annual leave per calendar year. Employees at Grade 7 and above accrue 28 days. Accrual begins on the first day of employment and is credited monthly in arrears.\nA maximum of 5 unused annual leave days may be carried into the following year. Carried days expire on 31 March and are not paid out on expiry."]} +{"id": "u08_bereavement_swapped", "label_grounded": false, "corpus": "hr", "failure_mode": "swapped entity", "question": "How much bereavement leave is there?", "answer": "Bereavement leave is 2 working days for an immediate family member and 5 working days for an extended family member.", "evidence": ["5. BEREAVEMENT LEAVE\nBereavement leave is 5 working days for an immediate family member and 2 working days for an extended family member.\n6. JURY DUTY\nJury service is paid in full for up to 15 working days per calendar year. Any court allowance received must be surrendered to the payroll team."]} +{"id": "u09_parental_transfer_added", "label_grounded": false, "corpus": "hr", "failure_mode": "unsupported addition", "question": "How much parental leave is there and how does it work?", "answer": "Parental leave is 18 weeks per child, of which 6 weeks are fully paid, and any unused weeks can be transferred to the other parent.", "evidence": ["4. PARENTAL LEAVE\nParental leave is 18 weeks per child, of which 6 weeks are fully paid. It must be taken before the child's third birthday. Parental leave may be split into no more than 3 separate blocks."]} +{"id": "u10_sabbatical_notice_wrong", "label_grounded": false, "corpus": "hr", "failure_mode": "wrong number", "question": "How much notice do I have to give for an unpaid sabbatical?", "answer": "Applications require 30 days written notice and are approved by the Head of People Operations.", "evidence": ["7. UNPAID SABBATICAL\nEmployees with at least 4 years of continuous service may apply for an unpaid sabbatical of up to 90 days. Applications require 60 days written notice and are approved by the Head of People Operations. A sabbatical does not interrupt continuous-service accrual."]} diff --git a/eval/multiturn/decomp_stability.py b/eval/multiturn/decomp_stability.py new file mode 100644 index 00000000..b54f7ad0 --- /dev/null +++ b/eval/multiturn/decomp_stability.py @@ -0,0 +1,203 @@ +"""Decomposition stability dump โ€” the single-turn byte-identity gate. + +The cheap first check from the "Rules going forward" in +eval/decisions/multiturn-decomposer-2026-08-16.md: the single-turn decomposer +prompt is a frozen measured artifact (arm L measured even cosmetic edits +shifting temp-0 decompositions), so ANY byte change to it must show up here +before it costs a full 5-bench gate. + +Dumps the temp-0 decompositions of all 120 single-turn gold queries +(goldset/{atlas7,hr,docs,acquisition,rfc}.jsonl, 24 rows each) through +``QueryDecomposer.decompose(query, [])`` โ€” empty history, i.e. the frozen +single-turn prompt path; temperature 0 is pinned in the decomposer itself. +The dump is perfectly deterministic run-to-run (120/120 byte-identical on a +repeat run, per the decision doc), so a diff against the baseline is a real +effect of a prompt or model change, never noise. + +Each dump row: {id, corpus, query, resolved_query, sub_queries}. +``resolved_query`` is the single sub-query when there is exactly one โ€” the +prompt's output rule 2 guarantees that entry IS the resolved query โ€” else +null. + +Usage (see eval/README.md): + + # (re)generate the baseline (the original scratchpad one is lost โ€” + # the first --write-baseline run recreates it) + .venv/bin/python eval/multiturn/decomp_stability.py --write-baseline + + # the gate: byte-compare a fresh dump against the baseline + .venv/bin/python eval/multiturn/decomp_stability.py --check + +Both flags take an optional path (default eval/multiturn/decomp_dump_pre.jsonl). +Needs Ollama running with the enrichment model; exits 2 when it cannot be +reached. Exit status of --check is 0 iff the fresh dump is byte-identical to +the baseline, 1 otherwise (differing ids are listed). +""" + +import argparse +import json +import os +import sys + +MT_DIR = os.path.dirname(os.path.abspath(__file__)) +EVAL_DIR = os.path.abspath(os.path.join(MT_DIR, "..")) +REPO_ROOT = os.path.abspath(os.path.join(EVAL_DIR, "..")) +sys.path.insert(0, REPO_ROOT) +sys.path.insert(0, EVAL_DIR) + +import requests # noqa: E402 + +import run_eval # noqa: E402 +from rag_system.factory import _build_llm_client # noqa: E402 +from rag_system.retrieval.query_transformer import QueryDecomposer # noqa: E402 + +# The five single-turn gold sets of record, 24 rows each = 120 queries. +SINGLE_TURN_GOLDSETS = ("atlas7", "hr", "docs", "acquisition", "rfc") +DEFAULT_DUMP_PATH = os.path.join(MT_DIR, "decomp_dump_pre.jsonl") + + +# -------------------------------------------------------------------------- +# dump +# -------------------------------------------------------------------------- + +def load_single_turn_queries() -> list: + """All 120 single-turn gold rows, sorted by id for a canonical dump order.""" + rows = [row for name in SINGLE_TURN_GOLDSETS + for row in run_eval._read_gold_file(name)] + return sorted(rows, key=lambda r: r["id"]) + + +def build_decomposer() -> tuple: + """The decomposer exactly as run_eval constructs it (decomposer = utility model).""" + llm_client, llm_config = _build_llm_client() + model = llm_config.get("enrichment_model") or llm_config["generation_model"] + return QueryDecomposer(llm_client, model), llm_config, model + + +def dump_records(decomposer: QueryDecomposer, rows: list) -> list: + """Temp-0 single-turn decomposition of every gold row, as dump records.""" + records = [] + for row in rows: + # Empty chat history -> the frozen single-turn prompt path. + sub_queries = decomposer.decompose(row["query"], [], max_sub_queries=10) + records.append({ + "id": row["id"], + "corpus": row.get("corpus"), + "query": row["query"], + # Output rule 2: no decomposition => the one sub-query IS the + # resolved query. With >1 sub-queries the resolved form is the + # decomposer's internal intermediate, so record null rather than + # guess. + "resolved_query": sub_queries[0] if len(sub_queries) == 1 else None, + "sub_queries": sub_queries, + }) + return records + + +def serialize(records: list) -> bytes: + """Canonical dump bytes: sorted keys, one JSON object per line.""" + return "".join(json.dumps(r, ensure_ascii=False, sort_keys=True) + "\n" + for r in records).encode("utf-8") + + +def parse_dump(data: bytes) -> dict: + """id -> record, for the human-readable diff report after a byte mismatch.""" + out = {} + for line in data.decode("utf-8").splitlines(): + if line.strip(): + record = json.loads(line) + out[record["id"]] = record + return out + + +def require_ollama(llm_config: dict, models: list) -> None: + """Probe Ollama before doing any work; exit 2 when it is not usable. + + ``OllamaClient`` swallows connection errors into ``{}``, and the + decomposer's fail-open then silently returns the raw query โ€” a dead server + would otherwise WRITE a degraded baseline with no error at all. + """ + host = llm_config["host"] + needed = sorted(set(models)) + try: + resp = requests.get(f"{host}/api/tags", timeout=5) + resp.raise_for_status() + available = {m.get("name", "") for m in resp.json().get("models", [])} + except Exception: + print(f"ERROR: cannot reach Ollama at {host} โ€” this gate needs Ollama " + f"running with: {', '.join(needed)}.", file=sys.stderr) + sys.exit(2) + missing = [m for m in needed + if m not in available and f"{m}:latest" not in available] + if missing: + print(f"ERROR: Ollama at {host} has no model(s): {', '.join(missing)} โ€” " + f"this gate needs: {', '.join(needed)} (`ollama pull` the missing " + f"ones first).", file=sys.stderr) + sys.exit(2) + + +# -------------------------------------------------------------------------- +# main +# -------------------------------------------------------------------------- + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + mode = parser.add_mutually_exclusive_group(required=True) + mode.add_argument("--write-baseline", nargs="?", const=DEFAULT_DUMP_PATH, + default=None, metavar="PATH", + help="dump decompositions to PATH " + "(default: eval/multiturn/decomp_dump_pre.jsonl)") + mode.add_argument("--check", nargs="?", const=DEFAULT_DUMP_PATH, + default=None, metavar="PATH", + help="byte-compare a fresh dump against the baseline at PATH " + "(default: eval/multiturn/decomp_dump_pre.jsonl)") + args = parser.parse_args() + + rows = load_single_turn_queries() + decomposer, llm_config, model = build_decomposer() + require_ollama(llm_config, [model]) + + os.makedirs(run_eval.RESULTS_DIR, exist_ok=True) + log_path = os.path.join(run_eval.RESULTS_DIR, "decomp_stability.log") + print(f"decomposing {len(rows)} single-turn gold queries with {model} " + f"(temp 0, frozen single-turn prompt)โ€ฆ") + with run_eval.captured(log_path, verbose=False): + records = dump_records(decomposer, rows) + dump = serialize(records) + + if args.write_baseline is not None: + path = args.write_baseline + with open(path, "wb") as fh: + fh.write(dump) + print(f"wrote {len(records)} decompositions to {path}") + return 0 + + path = args.check + try: + with open(path, "rb") as fh: + baseline = fh.read() + except FileNotFoundError: + print(f"ERROR: baseline {path} not found โ€” generate it first with " + f"--write-baseline (the original scratchpad baseline is lost).", + file=sys.stderr) + return 1 + + if dump == baseline: + print(f"identical: {len(records)}/{len(records)} decompositions " + f"byte-identical to {path}") + return 0 + + # Byte mismatch: parse only now, to name the differing rows. + base_by_id = parse_dump(baseline) + fresh_by_id = parse_dump(dump) + diff_ids = sorted({i for i in fresh_by_id if base_by_id.get(i) != fresh_by_id[i]} + | {i for i in base_by_id if i not in fresh_by_id}) + print(f"DIFF: {len(diff_ids)}/{len(records)} decompositions differ from {path}:") + for i in diff_ids: + print(f" {i}") + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/multiturn/run_e2e_multiturn.py b/eval/multiturn/run_e2e_multiturn.py new file mode 100644 index 00000000..32ff2f51 --- /dev/null +++ b/eval/multiturn/run_e2e_multiturn.py @@ -0,0 +1,246 @@ +"""Multi-turn end-to-end gate for the conversational decomposer (arms m0โ€“m1d). + +The runnable half of the "Rules going forward" in +eval/decisions/multiturn-decomposer-2026-08-16.md: any change to the +multi-turn decomposer prompt gates on eval/goldset/multiturn.jsonl here, plus +the single-turn byte-identity check in eval/multiturn/decomp_stability.py. + +Per the decision doc's contract, each conversation is executed sequentially +through ``Agent.run`` with a real ``session_id`` โ€” turn 2 sees whatever the +system actually answered (in-process chat history plus the session-scoped +semantic cache) โ€” and ONLY the final turn's answer is graded: case-insensitive +substring check of the row's ``expected`` strings, "any" (at least one +present) or "all" (every one) per the row's ``match`` field. One Agent serves +the whole run (``table_name`` is passed per ``Agent.run`` call); each +conversation gets a fresh ``session_id`` so history never leaks across rows. + +The index per corpus is the SAME cached LanceDB index the single-turn harness +builds (``run_eval.ensure_index`` and its fingerprint), and the config is the +shipped "default" profile via ``run_eval.build_config`` โ€” reranker ON with +arm-G threshold selection โ€” with one deliberate exception: ``build_config`` +disables ``query_decomposition`` because ``run_eval`` never calls +``Agent.run()``. This gate does, so the profile's decomposition block (arm H: +enabled, pooled_first_stage) is restored. Everything else ``build_config`` +turns off (enrichment, latechunk, context expansion, verification) stays off: +those are index-shape decisions the cached indexes were built to match. + +Usage (see eval/README.md): + + .venv/bin/python eval/multiturn/run_e2e_multiturn.py --json-out eval/results/mt_answers.jsonl + .venv/bin/python eval/multiturn/run_e2e_multiturn.py --ids mt_01,mt_07 --verbose + +Needs Ollama running (triage, decomposition and synthesis are LLM calls); +exits 2 naming the required models when Ollama cannot be reached or does not +have them. Exit status is 0 iff every selected row passes, 1 otherwise. +""" + +import argparse +import json +import os +import sys +import types +import uuid +from datetime import datetime, timezone + +MT_DIR = os.path.dirname(os.path.abspath(__file__)) +EVAL_DIR = os.path.abspath(os.path.join(MT_DIR, "..")) +REPO_ROOT = os.path.abspath(os.path.join(EVAL_DIR, "..")) +sys.path.insert(0, REPO_ROOT) +sys.path.insert(0, EVAL_DIR) + +import httpx # noqa: E402 +import requests # noqa: E402 + +import run_eval # noqa: E402 +from rag_system.agent.loop import Agent # noqa: E402 +from rag_system.factory import _build_llm_client, get_pipeline_config # noqa: E402 +from rag_system.main import EXTERNAL_MODELS # noqa: E402 + +GOLDSET_NAME = "multiturn" +K = 20 # run_eval's default; also the shipped profile's retrieval_k +CHUNK_SIZE = 512 # what the HTTP path sends (run_eval default) + + +# -------------------------------------------------------------------------- +# helpers +# -------------------------------------------------------------------------- + +def grade_final_answer(answer: str, expected: list, match: str) -> bool: + """Case-insensitive substring check of the final answer, "any"/"all". + + Exactly ``run_eval.query_hit`` over a one-text list, so the gate grades + with the same normalisation the retrieval harness scores with. + """ + return bool(run_eval.query_hit([answer or ""], expected, match)) + + +def require_ollama(llm_config: dict, models: list) -> None: + """Probe Ollama before doing any work; exit 2 when it is not usable. + + ``OllamaClient`` swallows connection errors into ``{}`` on both its sync + and async paths, so a missing server or an unpulled model would otherwise + surface as silently degraded answers (the decomposer falls back to the + raw query), not as an error. + """ + host = llm_config["host"] + needed = sorted(set(models)) + try: + resp = requests.get(f"{host}/api/tags", timeout=5) + resp.raise_for_status() + available = {m.get("name", "") for m in resp.json().get("models", [])} + except Exception: + print(f"ERROR: cannot reach Ollama at {host} โ€” this gate needs Ollama " + f"running with: {', '.join(needed)}.", file=sys.stderr) + sys.exit(2) + missing = [m for m in needed + if m not in available and f"{m}:latest" not in available] + if missing: + print(f"ERROR: Ollama at {host} has no model(s): {', '.join(missing)} โ€” " + f"this gate needs: {', '.join(needed)} (`ollama pull` the missing " + f"ones first).", file=sys.stderr) + sys.exit(2) + + +def corpus_index(corpus: str, embedder: str, force: bool, log_path: str, + verbose: bool) -> tuple: + """Build or reuse the same cached index ``run_eval`` would; return (cfg, table). + + The config is ``build_config``'s shipped-default profile with the arm-H + decomposition block restored (see the module docstring), so the gate + measures what ships. ``ensure_index``'s fingerprint keys the cache, so a + cached index built by ``run_eval.py --corpus <corpus>`` is reused as-is. + """ + db_path = os.path.join(run_eval.INDEX_ROOT, run_eval.slug(embedder), + run_eval.corpus_slug(corpus)) + table = f"eval_{run_eval.corpus_slug(corpus)}" + # rerank_settings on the shipped profile: reranker ON, profile model. + rerank_enabled, reranker_name = run_eval.rerank_settings( + types.SimpleNamespace(no_rerank=False, reranker=None)) + cfg = run_eval.build_config(corpus, embedder, reranker_name, db_path, table, + K, CHUNK_SIZE, rerank_enabled) + cfg["query_decomposition"] = get_pipeline_config("default")["query_decomposition"] + run_eval.ensure_index(corpus, embedder, CHUNK_SIZE, cfg, db_path, table, + force, log_path, verbose) + return cfg, db_path, table + + +def point_agent_at(agent: Agent, db_path: str, table: str) -> None: + """Re-point the agent's pipeline at another corpus's LanceDB path + table. + + The pipeline caches the LanceDB manager (and the dense retriever bound to + it) on first use, so switching corpora means dropping both; they rebuild + lazily on the next query. Same reset pattern as + ``RetrievalPipeline.update_embedding_model``. ``storage`` is mutated in + place because ``pipeline.storage_config`` is a reference to the same dict. + """ + pipeline = agent.retrieval_pipeline + storage = pipeline.config["storage"] + storage.pop("db_path", None) + storage["lancedb_uri"] = db_path + storage["text_table_name"] = table + pipeline.db_manager = None + pipeline.dense_retriever = None + pipeline._dense_retriever_error = None + + +def run_row(agent: Agent, row: dict, table: str, log_path: str, + verbose: bool) -> dict: + """Execute the row's turns in one session; grade the final turn's answer.""" + session_id = f"eval-mt-{row['id']}-{uuid.uuid4().hex[:8]}" + result = None + for turn in row["turns"]: + with run_eval.captured(log_path, verbose): + result = agent.run(turn, table_name=table, session_id=session_id) + answer = (result or {}).get("answer", "") + passed = grade_final_answer(answer, row["expected"], row.get("match", "any")) + return {"id": row["id"], "corpus": row["corpus"], "class": row["class"], + "pass": passed, "final_answer": answer, "expected": row["expected"]} + + +# -------------------------------------------------------------------------- +# main +# -------------------------------------------------------------------------- + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--json-out", default=None, + help="write one JSONL result row per conversation to this path") + parser.add_argument("--ids", default=None, + help="comma-separated subset of conversation ids, e.g. mt_01,mt_07") + parser.add_argument("--verbose", action="store_true", + help="let the pipeline print to stdout and show final answers") + parser.add_argument("--force-reindex", action="store_true", + help="rebuild the cached corpus indexes even if fingerprints match") + args = parser.parse_args() + + rows = run_eval._read_gold_file(GOLDSET_NAME) + if args.ids: + wanted = [i.strip() for i in args.ids.split(",") if i.strip()] + by_id = {row["id"]: row for row in rows} + unknown = [i for i in wanted if i not in by_id] + if unknown: + parser.error(f"unknown id(s): {', '.join(unknown)} " + f"(have: {', '.join(sorted(by_id))})") + rows = [by_id[i] for i in wanted] + rows = sorted(rows, key=lambda r: r["id"]) + + run_eval.seed_everything() + os.makedirs(run_eval.RESULTS_DIR, exist_ok=True) + os.makedirs(run_eval.INDEX_ROOT, exist_ok=True) + stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + log_path = os.path.join(run_eval.RESULTS_DIR, f"multiturn_{stamp}.log") + + embedder = EXTERNAL_MODELS["embedding_model"] + llm_client, llm_config = _build_llm_client() + require_ollama(llm_config, [llm_config["generation_model"], + llm_config.get("enrichment_model") + or llm_config["generation_model"]]) + + print("localGPT multi-turn E2E gate") + print(f" goldset {GOLDSET_NAME}.jsonl โ€” {len(rows)} conversation(s)") + print(f" embedder {embedder}") + print(f" log {log_path}") + + by_corpus = {} + for row in rows: + by_corpus.setdefault(row["corpus"], []).append(row) + + agent = None + results = [] + for corpus in sorted(by_corpus): + print(f"\n=== corpus: {corpus} โ€” {run_eval.CORPORA[corpus]['label']}") + cfg, db_path, table = corpus_index(corpus, embedder, args.force_reindex, + log_path, args.verbose) + if agent is None: + # One Agent for the whole run; table_name goes per Agent.run call. + agent = Agent(pipeline_configs=cfg, llm_client=llm_client, + ollama_config=llm_config) + point_agent_at(agent, db_path, table) + for row in by_corpus[corpus]: + try: + record = run_row(agent, row, table, log_path, args.verbose) + except (requests.exceptions.RequestException, httpx.HTTPError) as e: + print(f"ERROR: Ollama became unreachable mid-run ({e}) โ€” needs " + f"Ollama running with {llm_config['generation_model']} / " + f"{llm_config.get('enrichment_model')}.", file=sys.stderr) + return 2 + results.append(record) + print(f" [{record['id']}] {'PASS' if record['pass'] else 'FAIL'}") + if args.verbose: + print(f" answer: {record['final_answer']}") + + passed = sum(1 for r in results if r["pass"]) + print(f"\n{passed}/{len(results)} passed") + + if args.json_out: + with open(args.json_out, "w", encoding="utf-8") as fh: + for record in sorted(results, key=lambda r: r["id"]): + fh.write(json.dumps(record, ensure_ascii=False) + "\n") + print(f"results {args.json_out}") + + return 0 if passed == len(results) else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/run_eval.py b/eval/run_eval.py new file mode 100644 index 00000000..6d604f03 --- /dev/null +++ b/eval/run_eval.py @@ -0,0 +1,1062 @@ +"""Retrieval metrics runner for the localGPT gold set (Phase 0.2). + +In-process: no HTTP server is started. The harness builds a LanceDB index with +the repo's own ``IndexingPipeline`` and queries it with the repo's own +``RetrievalPipeline`` components (``MultiVectorRetriever`` for the first stage, +``_get_ai_reranker()`` for the cross-encoder), so the numbers describe the +shipped retrieval path โ€” not a reimplementation of it. + +What it deliberately does NOT do: answer synthesis, context expansion, Provence +pruning, late chunking, contextual enrichment. Those are downstream of the two +metrics the roadmap says matter (first-stage recall@k, post-rerank nDCG@10) and +each one adds an LLM round-trip or a nondeterministic step. + +Metric definitions (binary relevance, answer-bearing text match): + + hit(chunk, query) A gold row carries one or more ``expected`` substrings. + A chunk is relevant when its text contains any of them + (whitespace-normalised, case-insensitive). + recall@k match="any": 1.0 when at least one of the top-k chunks is + relevant. match="all" (comparatives): 1.0 only when every + expected substring appears somewhere in the top-k union. + Reported as the mean over queries โ€” with one gold target + per query this is recall, hit-rate and success@k alike. + nDCG@10 DCG over binary per-chunk relevance of the top 10, divided + by the IDCG of the *same candidate set* re-sorted ideally. + A query whose candidate set contains no relevant chunk + scores 0, so a first-stage miss is never hidden. + +Two candidate lists are scored, not one: + + first stage ``retrieve_candidates()["first_stage"]`` โ€” the retriever's + own ordering (``recall@k``, ``ndcg10_first_stage``). + final ``retrieve_candidates()["documents"]`` โ€” post-rerank AND + post-cross-reference-hop, i.e. the list the answer stage + would actually see (``recall_final``, ``ndcg10_final``). + +The distinction exists because roadmap item 4.2's cross-reference hop *appends* +to ``documents`` and deliberately never mutates ``first_stage``: scoring only +the first stage reports a flat line for the hop no matter how well it works. +With reranking off and the hop off the two lists are the same object, and the +run asserts exactly that (``final == first_stage`` invariant, printed and +recorded in the results JSON). + +Gold relevance is text-based, never chunk-id-based, so the set survives +re-chunking and embedder swaps. + +Usage (see eval/README.md): + + .venv/bin/python eval/run_eval.py --corpus all # shipped defaults + EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B .venv/bin/python eval/run_eval.py --corpus all + +The rerank stage follows the shipped "default" profile, which has reranking ON +with threshold selection since arm G (2026-08-14): ``top_k: 10`` plus +``min_score: 0.5`` / ``min_keep: 3`` pruning of whatever the Qwen scorer marks +irrelevant to every query. The harness keeps that selection when the stage is +on, so the ``final`` metrics describe the list the answer stage actually sees. +Pass ``--reranker <model>`` to swap the model, ``--no-rerank`` to force the +stage off for the first-stage-only control arm. +""" + +import argparse +import contextlib +import glob +import io +import json +import math +import os +import platform +import random +import shutil +import sys +import time +from datetime import datetime, timezone + +EVAL_DIR = os.path.dirname(os.path.abspath(__file__)) +REPO_ROOT = os.path.abspath(os.path.join(EVAL_DIR, "..")) +sys.path.insert(0, REPO_ROOT) + +import numpy as np # noqa: E402 +import torch # noqa: E402 + +from rag_system.factory import _build_llm_client, get_pipeline_config # noqa: E402 +from rag_system.indexing.embedders import LanceDBManager # noqa: E402 +from rag_system.main import EXTERNAL_MODELS # noqa: E402 +from rag_system.pipelines.indexing_pipeline import IndexingPipeline # noqa: E402 +from rag_system.pipelines.retrieval_pipeline import RetrievalPipeline # noqa: E402 + +SEED = 20260808 +INDEX_ROOT = os.path.join(EVAL_DIR, ".eval_indexes") +RESULTS_DIR = os.path.join(EVAL_DIR, "results") +GOLDSET_DIR = os.path.join(EVAL_DIR, "goldset") + +# Index-time code version, mixed into the index fingerprint. The fingerprint +# otherwise covers only file size/mtime and flags โ€” NOT this repo's code โ€” so a +# change to anything that alters what lands in the index (chunker, converter, +# crossref stamping, normalizationโ€ฆ) must bump this, or cached indexes built by +# the old code are silently reused (the manual cache deletion in BASELINE.md ยง +# "Why the rebuild" is what this automates). +INDEX_CODE_VERSION = "2026-08-16.1" + +# Decomposition-prompt version, part of the sub-query cache key. Bump when the +# QueryDecomposer prompt or its parameters change; combined with the resolved +# model name in the cache filename, stale decompositions are then never +# silently reused across models or prompt edits. +SUBQUERY_PROMPT_VERSION = "2026-08-16.1" + +# Excluded from the docs corpus on purpose: these two files are the planning +# documents this harness is tracked in, so every eval-related edit would change +# the corpus and move the baseline. Nothing in the gold set anchors on them. +DOCS_EXCLUDE = {"improvement_plan.md", "research_roadmap.md"} + +CORPORA = { + "atlas7": { + "label": "Atlas-7 service manual (planted-fact PDF)", + "files": [os.path.join(EVAL_DIR, "corpora", "atlas7_service_manual.pdf")], + }, + "hr": { + "label": "Northwind leave policy (synthetic planted-fact PDF)", + "files": [os.path.join(EVAL_DIR, "corpora", "northwind_leave_policy.pdf")], + }, + "docs": { + "label": "localGPT Documentation/*.md (real heterogeneous corpus)", + "glob": os.path.join(REPO_ROOT, "Documentation", "*.md"), + }, + # Roadmap Phase 4 (4.1 / 4.2 / 4.3) needs what no other corpus here has: many + # documents that reference *each other*. Ten synthetic M&A documents whose + # every "Document: <Title>", "Exhibit X" and "Schedule N" pointer resolves + # inside the corpus. Its gold set carries the extra `requires_crossref` + # dimension โ€” the 4.2 metric. + "acq": { + "label": "acquisition deal room (10 interlinked synthetic M&A PDFs)", + "glob": os.path.join(EVAL_DIR, "corpora", "acquisition", "*.pdf"), + "goldset_of": ["acquisition"], + }, + # The cross-reference corpus with the documentation corpus as distractors. + # Deliberately NOT folded into `mixed`: `mixed` is the tracked Phase 0/1/2 + # baseline and must keep meaning the same thing across the whole roadmap. + "acq+docs": { + "label": "acquisition deal room + Documentation/*.md as distractors", + "glob": [os.path.join(EVAL_DIR, "corpora", "acquisition", "*.pdf"), + os.path.join(REPO_ROOT, "Documentation", "*.md")], + "goldset_of": ["acquisition", "docs"], + }, + # Real third-party documents whose naming and referencing conventions this + # project did not invent โ€” 23 IETF RFCs (the QUIC / HTTP-3 family), fetched + # verbatim from rfc-editor.org by corpora/rfc/download.py (plain text, 1.44 + # MiB). The honest test of the index-time crossref extractor: it resolves 0 + # of 1403 references here (corpora/rfc/MANIFEST.md, + # decisions/rfc-shakedown-2026-08-13.md). Deliberately NOT in `mixed`, for + # the same reason as `acq`. + "rfc": { + "label": "23 interlinked IETF RFCs, plain text (QUIC / HTTP-3 family)", + "glob": os.path.join(EVAL_DIR, "corpora", "rfc", "*.txt"), + }, + # Every corpus in one table. The two planted-fact PDFs are only a handful of + # chunks each, so in isolation k=20 sweeps the whole document and recall@k + # saturates at 1.0 by construction. Here their queries have to beat the + # documentation chunks as distractors, which is the number worth tracking. + "mixed": { + "label": "all three corpora in one table (planted-fact PDFs + docs as distractors)", + "files": [ + os.path.join(EVAL_DIR, "corpora", "atlas7_service_manual.pdf"), + os.path.join(EVAL_DIR, "corpora", "northwind_leave_policy.pdf"), + ], + "glob": os.path.join(REPO_ROOT, "Documentation", "*.md"), + "goldset_of": ["atlas7", "hr", "docs"], + }, +} + +RECALL_KS = (5, 10, 20) +NDCG_K = 10 + + +# -------------------------------------------------------------------------- +# helpers +# -------------------------------------------------------------------------- + +def seed_everything() -> None: + random.seed(SEED) + np.random.seed(SEED) + torch.manual_seed(SEED) + + +def norm(text: str) -> str: + return " ".join((text or "").split()).lower() + + +def slug(name: str) -> str: + return name.replace("/", "__").replace(":", "_") + + +def corpus_slug(corpus: str) -> str: + """Filesystem- and LanceDB-safe form of a corpus key (``acq+docs`` has a ``+``).""" + return corpus.replace("+", "_plus_") + + +def corpus_files(corpus: str) -> list: + spec = CORPORA[corpus] + files = list(spec.get("files", [])) + patterns = spec.get("glob", []) + if isinstance(patterns, str): + patterns = [patterns] + for pattern in patterns: + files.extend(f for f in glob.glob(pattern) + if os.path.basename(f) not in DOCS_EXCLUDE) + return sorted(files) + + +def _read_gold_file(name: str) -> list: + path = os.path.join(GOLDSET_DIR, f"{name}.jsonl") + if not os.path.exists(path): + return [] + rows = [] + with open(path, "r", encoding="utf-8") as fh: + for line in fh: + line = line.strip() + if line and not line.startswith("//"): + rows.append(json.loads(line)) + return rows + + +def load_goldset(corpus: str) -> list: + names = CORPORA[corpus].get("goldset_of", [corpus]) + rows = [row for name in names for row in _read_gold_file(name)] + return sorted(rows, key=lambda r: r["id"]) + + +@contextlib.contextmanager +def captured(log_path: str, verbose: bool): + """Send the pipeline's chatty stdout to a log file unless --verbose.""" + if verbose: + yield + return + buf = io.StringIO() + try: + with contextlib.redirect_stdout(buf): + yield + finally: + with open(log_path, "a", encoding="utf-8") as fh: + fh.write(buf.getvalue()) + + +# -------------------------------------------------------------------------- +# index build / reuse +# -------------------------------------------------------------------------- + +def build_config(corpus: str, embedder: str, reranker: str, db_path: str, table: str, + k: int, chunk_size: int, rerank_enabled: bool, + aggregate: str = "mean", overviews: bool = False) -> dict: + """The 'default' profile with every nondeterministic / LLM-dependent stage off.""" + cfg = get_pipeline_config("default") + cfg["storage"]["lancedb_uri"] = db_path + cfg["storage"]["text_table_name"] = table + cfg["embedding_model_name"] = embedder + cfg["chunker_mode"] = "docling" + cfg["chunking"] = {"chunk_size": chunk_size} + # Off for determinism and speed โ€” every one of these is an LLM round-trip + # per chunk, per document, or a second embedding pass. + cfg["contextual_enricher"] = {"enabled": False, "window_size": 1} + # Document overviews are off by default here for the same reason enrichment + # is: one LLM call per document. Roadmap item 4.3's prefilter cannot be + # measured without them, though โ€” it scores the query against the + # `.vectors.npz` sidecar that only an overview-enabled build writes. So + # `--overviews on` switches them back on and, critically, redirects the + # output *inside this corpus's eval index directory* rather than the repo's + # shared `index_store/overviews/`: the sidecar is then owned by the index, + # deleted with it, and can never be read by the wrong corpus. + # + # Overviews do not touch the `text` or `vector` columns โ€” `build_and_store` + # only appends a JSONL line โ€” so an overview-enabled index is chunk-for-chunk + # identical to one built without them. + if overviews: + cfg["overview"] = {"enabled": True, "embed": True} + cfg["overview_path"] = os.path.join(db_path, "overviews.jsonl") + else: + cfg["overview"] = {"enabled": False} + cfg["retrieval"]["latechunk"] = {"enabled": False} + cfg["retrieval"]["search_type"] = "hybrid" + # `enabled: False` keeps the *agent's* decomposition branch out of it โ€” this + # harness never calls Agent.run(). Sub-queries, when --decompose is passed, + # are handed straight to the rerank stage, which is where item 2.2 puts them. + cfg["query_decomposition"] = {"enabled": False, "rerank_aggregate": aggregate} + cfg["verification"] = {"enabled": False} + cfg["context_window_size"] = 0 + cfg["retrieval_k"] = k + # Keep the shipped profile's reranker block and override only the model: + # since arm G the profile selects (min_score / min_keep) and truncates + # (top_k), it does not just reorder, and the pipeline only applies that + # threshold selection when min_score is present. Replacing the block with a + # bare one would make the `final` metrics describe a reorder-only stack + # that no longer ships โ€” `final` is supposed to be the list the answer + # stage actually sees. + cfg["reranker"]["enabled"] = rerank_enabled + cfg["reranker"]["model_name"] = reranker + return cfg + + +def apply_retry_setting(cfg: dict, mode: str) -> dict: + """Force the evidence-sufficiency retry on/off, or leave the profile's value. + + Unlike every other stage this harness disables, the retry is *conditional*: + it only fires on queries whose first pass found weak evidence, so leaving it + at the profile default is what measures the shipped stack. ``--no-retry`` + produces the control arm. + """ + if mode == "profile": + return cfg + block = dict(cfg["retrieval"].get("retry") or {}) + block["enabled"] = (mode == "on") + cfg["retrieval"]["retry"] = block + return cfg + + +def apply_phase4_settings(cfg: dict, crossref_mode: str, overview_mode: str) -> dict: + """Force the two Phase-4 query-time flags, or leave the profile's values. + + Same shape as ``apply_retry_setting``: ``profile`` means "whatever + ``main.py`` says" (both are OFF there today), anything else is a forced arm + for an A/B. Written under ``retrieval.<key>`` because that is the container + ``RetrievalPipeline._merged_block`` reads the profile from; the + ``retrievers.<key>`` spelling is the API's runtime-override lane and is left + alone so a forced arm here cannot be silently overridden. + """ + if crossref_mode != "profile": + block = dict(cfg["retrieval"].get("crossref_hop") or {}) + block["enabled"] = (crossref_mode == "on") + cfg["retrieval"]["crossref_hop"] = block + + if overview_mode != "profile": + block = dict(cfg["retrieval"].get("overview_prefilter") or {}) + block["enabled"] = (overview_mode != "off") + if overview_mode in ("boost", "restrict"): + block["mode"] = overview_mode + cfg["retrieval"]["overview_prefilter"] = block + return cfg + + +def index_fingerprint(corpus: str, embedder: str, chunk_size: int, + overviews: bool = False) -> dict: + files = corpus_files(corpus) + fingerprint = { + "corpus": corpus, + "embedder": embedder, + "chunk_size": chunk_size, + "enrichment": False, + "latechunk": False, + # Vectors are L2-normalized at write time since the Phase 1 adoption + # (eval/DECISIONS.md). Carrying it here invalidates every index built + # before that, exactly once, instead of silently reusing them. + "normalized": True, + # Code, not data: size/mtime cannot see a chunker or crossref-stamping + # fix, so the fingerprint carries the harness's index-code version. + "index_code_version": INDEX_CODE_VERSION, + "files": [ + {"path": os.path.relpath(f, REPO_ROOT), "size": os.path.getsize(f), + "mtime": round(os.path.getmtime(f), 3)} + for f in files + ], + } + # Overviews leave the chunk index bit-identical, but they are what produces + # the `.vectors.npz` sidecar item 4.3's prefilter reads โ€” so a cached index + # built without them cannot serve an `--overviews on` run. Added as a key + # only when true, so every index cached before this flag existed keeps its + # fingerprint and is not needlessly rebuilt. + if overviews: + fingerprint["overviews"] = True + return fingerprint + + +def ensure_index(corpus: str, embedder: str, chunk_size: int, cfg: dict, db_path: str, + table: str, force: bool, log_path: str, verbose: bool, + overviews: bool = False) -> dict: + marker = os.path.join(db_path, f"{table}.built.json") + fingerprint = index_fingerprint(corpus, embedder, chunk_size, overviews) + + if not force and os.path.exists(marker): + with open(marker, "r", encoding="utf-8") as fh: + existing = json.load(fh) + if existing.get("fingerprint") == fingerprint: + print(f" reusing cached index {db_path}::{table} " + f"({existing['chunks']} chunks, built {existing['built_at']})") + return existing + + if os.path.isdir(db_path): + shutil.rmtree(db_path) + os.makedirs(db_path, exist_ok=True) + + files = corpus_files(corpus) + print(f" building index {db_path}::{table} from {len(files)} file(s)โ€ฆ") + llm_client, llm_config = _build_llm_client() + t0 = time.time() + with captured(log_path, verbose): + pipeline = IndexingPipeline(cfg, llm_client, llm_config) + pipeline.run(files) + build_seconds = time.time() - t0 + + with captured(log_path, verbose): + tbl = LanceDBManager(db_path=db_path).get_table(table) + chunks = len(tbl.to_pandas()) + + record = { + "fingerprint": fingerprint, + "chunks": chunks, + "build_seconds": round(build_seconds, 2), + "built_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + } + with open(marker, "w", encoding="utf-8") as fh: + json.dump(record, fh, indent=2) + print(f" built {chunks} chunks in {build_seconds:.1f}s") + return record + + +def chunk_texts(db_path: str, table: str, log_path: str, verbose: bool) -> list: + with captured(log_path, verbose): + tbl = LanceDBManager(db_path=db_path).get_table(table) + df = tbl.to_pandas() + return [norm(t) for t in df["text"].tolist()] + + +# -------------------------------------------------------------------------- +# metrics +# -------------------------------------------------------------------------- + +def relevance(chunk_text: str, expected: list) -> int: + n = norm(chunk_text) + return 1 if any(norm(e) in n for e in expected) else 0 + + +def query_hit(ranked_texts: list, expected: list, match: str) -> int: + joined = " || ".join(norm(t) for t in ranked_texts) + if match == "all": + return 1 if all(norm(e) in joined for e in expected) else 0 + return 1 if any(norm(e) in joined for e in expected) else 0 + + +def ndcg_at_k(rels: list, k: int) -> float: + top = rels[:k] + dcg = sum(r / math.log2(i + 2) for i, r in enumerate(top)) + ideal = sorted(rels, reverse=True)[:k] + idcg = sum(r / math.log2(i + 2) for i, r in enumerate(ideal)) + return (dcg / idcg) if idcg > 0 else 0.0 + + +SUBQUERY_CACHE = os.path.join(INDEX_ROOT, "_subqueries") + + +def decomposition_sub_queries(pipeline: RetrievalPipeline, gold: list, args, + log_path: str) -> dict: + """Sub-queries per gold row for the roadmap-2.2 A/B, or ``{}`` when off. + + `QueryDecomposer` is an LLM call, so the result is cached and reused: the + decomposition-on and decomposition-off arms must differ only in whether the + sub-queries are *used*, never in what they are. The cache filename names + the corpus, the resolved decomposer model and SUBQUERY_PROMPT_VERSION, so + switching ENRICHMENT_MODEL or editing the prompt can never silently reuse + another run's decompositions. The first stage never sees the sub-queries โ€” + item 2.2's whole point is that they apply at reranking โ€” which is also why + a rerank-less run skips the work entirely: they would have no consumer. + """ + if not args.decompose: + return {} + if not rerank_settings(args)[0]: + print(" decompose skipped โ€” rerank stage is off, sub-queries " + "would have no consumer") + return {} + os.makedirs(SUBQUERY_CACHE, exist_ok=True) + + from rag_system.retrieval.query_transformer import QueryDecomposer # noqa: E402 + llm_client, llm_config = _build_llm_client() + model = llm_config.get("enrichment_model") or llm_config["generation_model"] + decomposer = QueryDecomposer(llm_client, model) + + path = os.path.join( + SUBQUERY_CACHE, + f"{corpus_slug(args.corpus_being_run)}.{slug(model)}.{SUBQUERY_PROMPT_VERSION}.json") + cache = {} + if os.path.exists(path): + with open(path, "r", encoding="utf-8") as fh: + cache = json.load(fh) + + missing = [row for row in gold if row["id"] not in cache] + if missing: + print(f" decomposing {len(missing)} query/queries with {model}โ€ฆ") + for row in missing: + with captured(log_path, args.verbose): + cache[row["id"]] = decomposer.decompose(row["query"], [], max_sub_queries=10) + with open(path, "w", encoding="utf-8") as fh: + json.dump(cache, fh, indent=2) + multi = sum(1 for v in cache.values() if len(v) > 1) + print(f" sub-queries {multi}/{len(gold)} rows decomposed into >1 sub-query") + return cache + + +# -------------------------------------------------------------------------- +# main evaluation +# -------------------------------------------------------------------------- + +def rerank_settings(args) -> tuple: + """(enabled, model_name) for this run. + + The shipped "default" profile has reranking ON since arm G (2026-08-14), + with threshold selection (``min_score`` / ``min_keep`` / ``top_k``), and a + bare eval run must measure what ships โ€” including that selection, which + ``build_config`` preserves. Naming a model with ``--reranker`` keeps the + stage on but swaps the model, which is what the A/B commands in + eval/decisions/*.md do; ``--no-rerank`` always wins. + """ + profile_default = bool( + get_pipeline_config("default").get("reranker", {}).get("enabled", False) + ) + enabled = (not args.no_rerank) and (args.reranker is not None or profile_default) + return enabled, (args.reranker or EXTERNAL_MODELS["reranker_model"]) + + +def evaluate_corpus(corpus: str, args, log_path: str) -> dict: + embedder = args.embedder + rerank_enabled, reranker_name = rerank_settings(args) + # An overview-enabled build lives in its own directory rather than replacing + # the tracked one. The chunk index is identical either way (overviews only + # append a JSONL line), so sharing a directory would work โ€” but it would make + # every alternation between `--overviews off` and `--overviews on` a full + # re-index, and would silently blow away the index another run is reading. + index_dir = corpus_slug(corpus) + ("_ov" if args.overviews == "on" else "") + db_path = os.path.join(INDEX_ROOT, slug(embedder), index_dir) + table = f"eval_{corpus_slug(corpus)}" + + args.corpus_being_run = corpus + print(f"\n=== corpus: {corpus} โ€” {CORPORA[corpus]['label']}") + overviews = (args.overviews == "on") + cfg = build_config(corpus, embedder, reranker_name, db_path, table, + args.k, args.chunk_size, rerank_enabled, args.aggregate, + overviews=overviews) + cfg = apply_retry_setting(cfg, args.retry) + cfg = apply_phase4_settings(cfg, args.crossref_hop, args.overview_prefilter) + index_record = ensure_index(corpus, embedder, args.chunk_size, cfg, db_path, table, + args.force_reindex, log_path, args.verbose, + overviews=overviews) + + gold = load_goldset(corpus) + if not gold: + names = CORPORA[corpus].get("goldset_of", [corpus]) + print(f" no gold rows for {names} in {GOLDSET_DIR} โ€” skipping queries") + return {"corpus": corpus, "index": index_record, "queries": [], "coverage": []} + + # Gate 2 of gold verification: does the answer-bearing text survive + # conversion + chunking into at least one indexed chunk? A gold row that + # fails here is structurally unreachable and is reported, not hidden. + all_chunks = chunk_texts(db_path, table, log_path, args.verbose) + coverage = [] + for row in gold: + missing = [e for e in row["expected"] if not any(norm(e) in c for c in all_chunks)] + if missing: + coverage.append({"id": row["id"], "missing": missing}) + if coverage: + print(f" โš ๏ธ {len(coverage)} gold row(s) whose expected text is in NO indexed chunk:") + for c in coverage: + print(f" {c['id']}: {c['missing']}") + else: + print(f" gold coverage {len(gold)}/{len(gold)} rows reachable in the index") + + if args.coverage_only: + return {"corpus": corpus, "index": index_record, "queries": [], "coverage": coverage} + + with captured(log_path, args.verbose): + pipeline = RetrievalPipeline(cfg, *_build_llm_client()) + pipeline.retriever # force lazy init outside the timed section + + decompose = decomposition_sub_queries(pipeline, gold, args, log_path) + + results = [] + for row in gold: + query, expected, match = row["query"], row["expected"], row.get("match", "any") + + # Go through the pipeline's own candidate path (first stage + optional + # rerank + the evidence-sufficiency retry) rather than calling the + # retriever directly, so the harness measures the shipped behaviour. + t0 = time.time() + with captured(log_path, args.verbose): + out = pipeline.retrieve_candidates(query, table_name=table, + sub_queries=decompose.get(row["id"])) + total_ms = (time.time() - t0) * 1000.0 + docs = out["first_stage"] + first_texts = [d.get("text", "") for d in docs] + + # The list the answer stage would actually see: post-rerank AND + # post-crossref-hop. Scoring only `first_stage` is what made roadmap 4.2 + # unmeasurable โ€” the hop appends to `documents` and never touches + # `first_stage` by design (retrieval_pipeline.py `_crossref_hop`). + final_docs = out.get("documents") or [] + final_texts = [d.get("text", "") for d in final_docs] + # The hop only ever *appends*, so removing the tagged rows reconstructs + # the post-rerank / pre-hop list exactly. That keeps `ndcg10_reranked` + # meaning what it has always meant across every earlier decision file. + pre_hop_docs = [d for d in final_docs if not d.get("via_crossref")] + hopped_docs = [d for d in final_docs if d.get("via_crossref")] + + entry = { + "id": row["id"], + "query": query, + "dimensions": row.get("dimensions", {}), + "match": match, + "expected": expected, + "candidates": len(docs), + "final_candidates": len(final_docs), + "first_stage_ms": round(total_ms, 1), + "recall": {f"@{k}": query_hit(first_texts[:k], expected, match) for k in RECALL_KS}, + "ndcg10_first_stage": round( + ndcg_at_k([relevance(t, expected) for t in first_texts], NDCG_K), 4), + "recall_final": {f"@{k}": query_hit(final_texts[:k], expected, match) + for k in RECALL_KS}, + "ndcg10_final": round( + ndcg_at_k([relevance(t, expected) for t in final_texts], NDCG_K), 4), + } + # The flags-off invariant, checked per query rather than argued: with no + # rerank and no hop the final list must BE the first-stage list. + entry["final_equals_first_stage"] = ( + [d.get("chunk_id") for d in final_docs] == [d.get("chunk_id") for d in docs]) + + # Hop instrumentation: rank movement alone cannot say whether the hop + # pulled the *right* document, so record both what it pulled and + # whether any of it was on target. + entry["crossref_chunks_in_final"] = len(hopped_docs) + if hopped_docs: + expected_sources = row.get("expected_sources") or [] + hop_docs = sorted({d.get("document_id") for d in hopped_docs if d.get("document_id")}) + entry["crossref_documents"] = hop_docs + # precision, document level: did a hop land in a gold source document? + entry["crossref_hit_expected_source"] = bool( + expected_sources and any(d in expected_sources for d in hop_docs)) + # precision, text level: does a hopped chunk actually carry gold text? + entry["crossref_chunk_relevant"] = bool( + any(relevance(d.get("text", ""), expected) for d in hopped_docs)) + # rank of the first relevant hopped chunk inside the final list + rels = [i for i, t in enumerate(final_texts) if relevance(t, expected)] + entry["first_relevant_rank_final"] = (rels[0] + 1) if rels else None + if out.get("crossref_hop"): + entry["crossref_hop"] = out["crossref_hop"] + if out.get("retry"): + entry["retry"] = out["retry"] + if decompose.get(row["id"]): + entry["sub_queries"] = decompose[row["id"]] + + if rerank_enabled and docs: + entry["rerank_ms"] = 0.0 # folded into first_stage_ms on this path + # Identity, not equality: the rerank stage signals failure by + # handing its input straight back. (The hop rebuilds `documents` as + # a new list, so `is` on the list itself is no longer sufficient โ€” + # the element identities are.) + rerank_noop = (len(pre_hop_docs) == len(docs) + and all(a is b for a, b in zip(pre_hop_docs, docs))) + if not pre_hop_docs or rerank_noop: + entry["rerank_error"] = "reranker failed to load; see the run log" + else: + rr_texts = [d.get("text", "") for d in pre_hop_docs] + entry["ndcg10_reranked"] = round( + ndcg_at_k([relevance(t, expected) for t in rr_texts], NDCG_K), 4) + entry["recall_reranked"] = { + f"@{k}": query_hit(rr_texts[:k], expected, match) for k in RECALL_KS + } + results.append(entry) + rr_val = entry.get("ndcg10_reranked") + rr_cell = "n/a " if rr_val is None else f"{rr_val:.3f}" + hop_cell = (f" hop=+{entry['crossref_chunks_in_final']}" + f"{'โœ“' if entry.get('crossref_chunk_relevant') else ''}" + if entry["crossref_chunks_in_final"] else "") + print(f" [{entry['id']}] r@5={entry['recall']['@5']} r@10={entry['recall']['@10']} " + f"r@20={entry['recall']['@20']} " + f"nDCG@10 first={entry['ndcg10_first_stage']:.3f} rerank={rr_cell} " + f"final={entry['ndcg10_final']:.3f}{hop_cell} " + f"{entry['first_stage_ms']:.0f}ms+{entry.get('rerank_ms', 0):.0f}ms") + + return {"corpus": corpus, "index": index_record, "queries": results, "coverage": coverage} + + +def mean(values: list) -> float: + return sum(values) / len(values) if values else float("nan") + + +def summarise(corpus_result: dict) -> dict: + qs = corpus_result["queries"] + if not qs: + return {} + summary = { + "n_queries": len(qs), + "recall@5": round(mean([q["recall"]["@5"] for q in qs]), 4), + "recall@10": round(mean([q["recall"]["@10"] for q in qs]), 4), + "recall@20": round(mean([q["recall"]["@20"] for q in qs]), 4), + "ndcg@10_first_stage": round(mean([q["ndcg10_first_stage"] for q in qs]), 4), + # Final = the list the answer stage sees: post-rerank, post-crossref-hop. + "recall@5_final": round(mean([q["recall_final"]["@5"] for q in qs]), 4), + "recall@10_final": round(mean([q["recall_final"]["@10"] for q in qs]), 4), + "recall@20_final": round(mean([q["recall_final"]["@20"] for q in qs]), 4), + "ndcg@10_final": round(mean([q["ndcg10_final"] for q in qs]), 4), + "final_equals_first_stage": all(q.get("final_equals_first_stage") for q in qs), + "first_stage_ms_mean": round(mean([q["first_stage_ms"] for q in qs]), 1), + "first_stage_ms_p90": round(sorted(q["first_stage_ms"] for q in qs)[int(0.9 * (len(qs) - 1))], 1), + } + fired = [q for q in qs if q.get("retry")] + if fired: + kept = [q for q in fired if q["retry"]["kept"] == "retry"] + summary["retry_fired"] = len(fired) + summary["retry_fire_rate"] = round(len(fired) / len(qs), 4) + summary["retry_kept"] = len(kept) + hopped = [q for q in qs if q.get("crossref_chunks_in_final")] + if hopped: + summary["crossref_hop"] = _hop_summary(qs, hopped) + reranked = [q for q in qs if "ndcg10_reranked" in q] + if reranked: + summary["ndcg@10_reranked"] = round(mean([q["ndcg10_reranked"] for q in reranked]), 4) + summary["rerank_ms_mean"] = round(mean([q["rerank_ms"] for q in reranked]), 1) + summary["rerank_ms_p90"] = round( + sorted(q["rerank_ms"] for q in reranked)[int(0.9 * (len(reranked) - 1))], 1) + # Roadmap 4.2's metric: the rows whose answer text lives in a document other + # than the one the query's premise points at. Reported next to the whole-corpus + # numbers rather than only in `by_dimension`, because the point of the slice is + # the *gap* between it and the rest of the same corpus. + crossref = [q for q in qs if q["dimensions"].get("requires_crossref") is True] + if crossref: + rest = [q for q in qs if q["dimensions"].get("requires_crossref") is False] + summary["crossref"] = _slice_summary(crossref) + if rest: + summary["crossref_control"] = _slice_summary(rest) + return summary + + +def _hop_summary(qs: list, hopped: list) -> dict: + """Did the hop fire, and did what it pulled belong there? + + ``fire_rate`` is over *all* queries in the slice; the precision numbers are + over the queries that actually hopped, because a query that never hopped is + not evidence about hop precision either way. + """ + return { + "queries_with_hop": len(hopped), + "fire_rate": round(len(hopped) / len(qs), 4) if qs else 0.0, + "chunks_added_total": sum(q["crossref_chunks_in_final"] for q in hopped), + "chunks_added_mean_when_fired": round( + mean([q["crossref_chunks_in_final"] for q in hopped]), 2), + # document-level precision: the hop landed in a gold source document + "hit_expected_source": sum(1 for q in hopped + if q.get("crossref_hit_expected_source")), + # text-level precision: a hopped chunk actually carries the gold text + "hopped_chunk_relevant": sum(1 for q in hopped + if q.get("crossref_chunk_relevant")), + } + + +def _slice_summary(qs: list) -> dict: + out = { + "n_queries": len(qs), + "recall@5": round(mean([q["recall"]["@5"] for q in qs]), 4), + "recall@10": round(mean([q["recall"]["@10"] for q in qs]), 4), + "recall@20": round(mean([q["recall"]["@20"] for q in qs]), 4), + "ndcg@10_first_stage": round(mean([q["ndcg10_first_stage"] for q in qs]), 4), + "recall@5_final": round(mean([q["recall_final"]["@5"] for q in qs]), 4), + "recall@10_final": round(mean([q["recall_final"]["@10"] for q in qs]), 4), + "recall@20_final": round(mean([q["recall_final"]["@20"] for q in qs]), 4), + "ndcg@10_final": round(mean([q["ndcg10_final"] for q in qs]), 4), + } + reranked = [q for q in qs if "ndcg10_reranked" in q] + if reranked: + out["ndcg@10_reranked"] = round(mean([q["ndcg10_reranked"] for q in reranked]), 4) + hopped = [q for q in qs if q.get("crossref_chunks_in_final")] + if hopped: + out["crossref_hop"] = _hop_summary(qs, hopped) + return out + + +# `requires_crossref` only exists on the `acq` gold set; rows without it are +# skipped, so adding the corpus cannot move any pre-existing slice. +DIMENSION_KEYS = ("question_type", "difficulty", "requires_crossref") + + +def by_dimension(all_results: list) -> dict: + buckets = {} + for corpus_result in all_results: + for q in corpus_result["queries"]: + for key in DIMENSION_KEYS: + value = q["dimensions"].get(key) + if value is None: + continue + if isinstance(value, bool): + value = "true" if value else "false" + bucket = buckets.setdefault(f"{key}={value}", + {"n": 0, "r@10": 0.0, "ndcg1": 0.0, + "ndcg": 0.0, "ndcg_n": 0, + "r@10f": 0.0, "ndcgf": 0.0, "hops": 0}) + bucket["n"] += 1 + bucket["r@10"] += q["recall"]["@10"] + bucket["ndcg1"] += q["ndcg10_first_stage"] + bucket["r@10f"] += q["recall_final"]["@10"] + bucket["ndcgf"] += q["ndcg10_final"] + if q.get("crossref_chunks_in_final"): + bucket["hops"] += 1 + if "ndcg10_reranked" in q: + bucket["ndcg"] += q["ndcg10_reranked"] + bucket["ndcg_n"] += 1 + for bucket in buckets.values(): + bucket["recall@10"] = round(bucket.pop("r@10") / bucket["n"], 4) + bucket["ndcg@10_first_stage"] = round(bucket.pop("ndcg1") / bucket["n"], 4) + bucket["recall@10_final"] = round(bucket.pop("r@10f") / bucket["n"], 4) + bucket["ndcg@10_final"] = round(bucket.pop("ndcgf") / bucket["n"], 4) + bucket["queries_with_crossref_hop"] = bucket.pop("hops") + ndcg_n = bucket.pop("ndcg_n") + total = bucket.pop("ndcg") + bucket["ndcg@10_reranked"] = round(total / ndcg_n, 4) if ndcg_n else None + return dict(sorted(buckets.items())) + + +def print_table(all_results: list) -> None: + header = (f"{'corpus':<12} {'n':>4} {'chunks':>7} {'R@5':>7} {'R@10':>7} {'R@20':>7} " + f"{'nDCG@10':>9} {'nDCG@10':>9} | {'R@5':>7} {'R@10':>7} {'R@20':>7} " + f"{'nDCG@10':>9} {'hop q':>6} {'1st ms':>8}") + print("\n" + "=" * len(header)) + print(header) + print(f"{'':<12} {'':>4} {'':>7} {'':>7} {'':>7} {'':>7} {'(1st)':>9} {'(rerank)':>9} | " + f"{'(fin)':>7} {'(fin)':>7} {'(fin)':>7} {'(final)':>9} {'':>6} {'':>8}") + print("-" * len(header)) + for corpus_result in all_results: + s = summarise(corpus_result) + if not s: + continue + rr = s.get("ndcg@10_reranked") + rr_cell = "n/a" if rr is None else f"{rr:.3f}" + hop = s.get("crossref_hop") or {} + hop_cell = str(hop.get("queries_with_hop", 0)) + print(f"{corpus_result['corpus']:<12} {s['n_queries']:>4} " + f"{corpus_result['index']['chunks']:>7} " + f"{s['recall@5']:>7.3f} {s['recall@10']:>7.3f} {s['recall@20']:>7.3f} " + f"{s['ndcg@10_first_stage']:>9.3f} {rr_cell:>9} | " + f"{s['recall@5_final']:>7.3f} {s['recall@10_final']:>7.3f} " + f"{s['recall@20_final']:>7.3f} {s['ndcg@10_final']:>9.3f} {hop_cell:>6} " + f"{s['first_stage_ms_mean']:>8.0f}") + for label, key in ((" โ”œ crossref", "crossref"), (" โ”” control ", "crossref_control")): + sl = s.get(key) + if not sl: + continue + sl_rr = sl.get("ndcg@10_reranked") + sl_hop = sl.get("crossref_hop") or {} + print(f"{label:<12} {sl['n_queries']:>4} {'':>7} " + f"{sl['recall@5']:>7.3f} {sl['recall@10']:>7.3f} {sl['recall@20']:>7.3f} " + f"{sl['ndcg@10_first_stage']:>9.3f} " + f"{('n/a' if sl_rr is None else f'{sl_rr:.3f}'):>9} | " + f"{sl['recall@5_final']:>7.3f} {sl['recall@10_final']:>7.3f} " + f"{sl['recall@20_final']:>7.3f} {sl['ndcg@10_final']:>9.3f} " + f"{str(sl_hop.get('queries_with_hop', 0)):>6} {'':>8}") + print("=" * len(header)) + print("left of the bar: first stage (and rerank). right of the bar: the FINAL " + "candidate list\n(post-rerank, post-crossref-hop) โ€” what the answer stage " + "would actually see.") + + +def check_final_invariant(all_results: list, rerank_enabled: bool, hop_desc: str) -> dict: + """With reranking and the hop both off, the final list must BE the first stage. + + This is the guard that makes the new ``*_final`` metrics trustworthy: if + they ever diverge from the first-stage metrics on a run where nothing is + allowed to reorder or append, the metric is measuring its own bug rather + than the pipeline. + """ + applicable = (not rerank_enabled) and hop_desc.startswith("off") + offenders = [] + for corpus_result in all_results: + for q in corpus_result["queries"]: + same_list = q.get("final_equals_first_stage") + same_metrics = (q["ndcg10_final"] == q["ndcg10_first_stage"] + and q["recall_final"] == q["recall"]) + if not (same_list and same_metrics): + offenders.append(f"{corpus_result['corpus']}/{q['id']}") + report = {"applicable": applicable, "offenders": offenders, + "checked": sum(len(r["queries"]) for r in all_results)} + if not applicable: + print(f"\ninvariant final == first_stage: not applicable " + f"(rerank={'on' if rerank_enabled else 'off'}, crossref hop={hop_desc}); " + f"{len(offenders)}/{report['checked']} queries have a final list that " + f"differs from the first stage.") + return report + if offenders: + print(f"\ninvariant โŒ FAIL โ€” final != first_stage on " + f"{len(offenders)}/{report['checked']} queries with rerank OFF and hop OFF: " + f"{', '.join(offenders[:10])}") + else: + print(f"\ninvariant โœ… final == first_stage on all {report['checked']} queries " + f"(rerank OFF, crossref hop OFF) โ€” chunk-id order and both metrics") + return report + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--corpus", default="all", choices=["all", *sorted(CORPORA)], + help="corpus to evaluate (default: all)") + parser.add_argument("--embedder", default=EXTERNAL_MODELS["embedding_model"], + help="HF embedding model; defaults to EMBEDDING_MODEL / the repo default") + parser.add_argument("--reranker", default=None, + help="reranker model name; defaults to the shipped profile's " + "reranker model (the stage itself follows the profile โ€” " + "ON since arm G โ€” unless --no-rerank is given)") + parser.add_argument("--k", type=int, default=20, help="first-stage candidates per query") + parser.add_argument("--chunk-size", type=int, default=512, + help="token budget per chunk (512 = what the HTTP path sends)") + parser.add_argument("--no-rerank", action="store_true", help="skip the cross-encoder stage") + parser.add_argument("--retry", choices=["profile", "on", "off"], default="profile", + help="evidence-sufficiency retry: follow the shipped profile " + "(default), or force it on/off for an A/B") + parser.add_argument("--crossref-hop", choices=["profile", "on", "off"], default="profile", + help="cross-reference hop (roadmap 4.2): follow the shipped " + "profile (default โ€” currently OFF), or force it on/off. " + "Only the FINAL metrics can move; the hop never touches " + "the first stage.") + parser.add_argument("--overview-prefilter", choices=["profile", "off", "boost", "restrict"], + default="profile", + help="overview prefilter (roadmap 4.3): follow the shipped " + "profile (default โ€” currently OFF), force it off, or " + "enable it in boost / restrict mode") + parser.add_argument("--overviews", choices=["off", "on"], default="off", + help="build per-document overviews at index time (one LLM " + "call per document) and the .vectors.npz sidecar the " + "overview prefilter reads. OFF by default, like every " + "other LLM stage here. Required for any " + "--overview-prefilter boost/restrict arm to do anything; " + "changes the index fingerprint, so it forces one rebuild. " + "Chunk text and vectors are unaffected.") + parser.add_argument("--aggregate", choices=["max", "mean"], default="mean", + help="how to combine per-sub-query rerank scores (--decompose only)") + parser.add_argument("--decompose", action="store_true", + help="decompose each query and score candidates against the " + "sub-queries AT RERANK (roadmap 2.2). The first stage always " + "uses the full original query. No-op without --reranker.") + parser.add_argument("--coverage-only", action="store_true", + help="build/reuse indexes and report gold reachability, then stop") + parser.add_argument("--force-reindex", action="store_true", help="rebuild even if cached") + parser.add_argument("--json-out", default=None, help="results path (default: eval/results/<ts>.json)") + parser.add_argument("--verbose", action="store_true", help="let the pipeline print to stdout") + args = parser.parse_args() + + seed_everything() + os.makedirs(RESULTS_DIR, exist_ok=True) + os.makedirs(INDEX_ROOT, exist_ok=True) + + stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + json_out = args.json_out or os.path.join(RESULTS_DIR, f"eval_{stamp}.json") + log_path = os.path.join(RESULTS_DIR, f"eval_{stamp}.log") + + corpora = sorted(CORPORA) if args.corpus == "all" else [args.corpus] + + rerank_enabled, reranker_name = rerank_settings(args) + + print("localGPT retrieval eval") + print(f" embedder {args.embedder}") + print(f" reranker {reranker_name if rerank_enabled else '(disabled)'}") + retry_cfg = (get_pipeline_config("default").get("retrieval", {}).get("retry") or {}) + if args.retry == "profile": + retry_desc = ("on (profile)" if retry_cfg.get("enabled") else "off (profile)") + else: + retry_desc = f"{args.retry} (forced)" + print(f" retry {retry_desc} min_top_score {retry_cfg.get('min_top_score')}") + profile_retrieval = get_pipeline_config("default").get("retrieval", {}) + hop_cfg = profile_retrieval.get("crossref_hop") or {} + pre_cfg = profile_retrieval.get("overview_prefilter") or {} + if args.crossref_hop == "profile": + hop_desc = ("on (profile)" if hop_cfg.get("enabled") else "off (profile)") + else: + hop_desc = f"{args.crossref_hop} (forced)" + if args.overview_prefilter == "profile": + pre_desc = (f"{pre_cfg.get('mode', 'boost')} (profile)" + if pre_cfg.get("enabled") else "off (profile)") + else: + pre_desc = f"{args.overview_prefilter} (forced)" + print(f" crossref {hop_desc} max_hops {hop_cfg.get('max_hops')} " + f"chunks_per_hop {hop_cfg.get('chunks_per_hop')}") + print(f" prefilter {pre_desc} top_documents {pre_cfg.get('top_documents')}") + if args.decompose and not rerank_enabled: + decompose_desc = "skipped โ€” rerank stage is off" + elif args.decompose: + decompose_desc = "on โ€” applied at rerank only, aggregate=" + args.aggregate + else: + decompose_desc = "off" + print(f" decompose {decompose_desc}") + print(f" k {args.k} chunk_size {args.chunk_size}") + print(f" enrichment OFF overviews {args.overviews.upper()} latechunk OFF " + f"context-expansion OFF") + print(f" log {log_path}") + + t0 = time.time() + all_results = [evaluate_corpus(c, args, log_path) for c in corpora] + wall_seconds = time.time() - t0 + + invariant = None + if not args.coverage_only: + print_table(all_results) + invariant = check_final_invariant(all_results, rerank_enabled, hop_desc) + + payload = { + "run": { + "timestamp": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "wall_seconds": round(wall_seconds, 1), + "seed": SEED, + "platform": f"{platform.system()} {platform.machine()} python {platform.python_version()}", + "torch": torch.__version__, + "device": ("cuda" if torch.cuda.is_available() + else "mps" if torch.backends.mps.is_available() else "cpu"), + "embedder": args.embedder, + "reranker": reranker_name if rerank_enabled else None, + "retry": retry_desc, + "retry_config": retry_cfg, + "crossref_hop": hop_desc, + "crossref_hop_config": hop_cfg, + "overview_prefilter": pre_desc, + "overview_prefilter_config": pre_cfg, + "decompose_at_rerank": bool(args.decompose) and rerank_enabled, + "rerank_aggregate": (args.aggregate + if args.decompose and rerank_enabled else None), + "k": args.k, + "chunk_size": args.chunk_size, + "enrichment": False, + "overviews": (args.overviews == "on"), + "latechunk": False, + "context_expansion": False, + "argv": sys.argv[1:], + }, + "summary": {r["corpus"]: summarise(r) for r in all_results if summarise(r)}, + "by_dimension": by_dimension(all_results), + # Same slices, not pooled across corpora โ€” a corpus that appears twice in + # one run (e.g. `acq` and `acq+docs`) would otherwise double-count its rows. + "by_dimension_per_corpus": {r["corpus"]: by_dimension([r]) + for r in all_results if r["queries"]}, + "final_vs_first_stage_invariant": invariant, + "coverage_failures": {r["corpus"]: r["coverage"] for r in all_results if r["coverage"]}, + "corpora": all_results, + } + with open(json_out, "w", encoding="utf-8") as fh: + json.dump(payload, fh, indent=2) + + print(f"\nwall clock {wall_seconds:.1f}s") + print(f"results {json_out}") + if payload["by_dimension"] and not args.coverage_only: + print("\nby dimension, pooled over the corpora in this run " + "(recall@10 / nDCG@10 first stage / nDCG@10 post-rerank / final):") + for key, bucket in payload["by_dimension"].items(): + nd = bucket["ndcg@10_reranked"] + print(f" {key:<28} n={bucket['n']:<3} recall@10={bucket['recall@10']:.3f} " + f"nDCG@10(1st)={bucket['ndcg@10_first_stage']:.3f} " + f"nDCG@10(rr)={'n/a' if nd is None else f'{nd:.3f}'} " + f"recall@10(fin)={bucket['recall@10_final']:.3f} " + f"nDCG@10(fin)={bucket['ndcg@10_final']:.3f}") + for corpus, buckets in payload["by_dimension_per_corpus"].items(): + crossref = {k: v for k, v in buckets.items() if k.startswith("requires_crossref=")} + if not crossref: + continue + print(f"\n {corpus} โ€” roadmap 4.2 slice:") + for key, bucket in sorted(crossref.items()): + print(f" {key:<26} n={bucket['n']:<3} recall@10={bucket['recall@10']:.3f} " + f"nDCG@10(1st)={bucket['ndcg@10_first_stage']:.3f} " + f"| recall@10(fin)={bucket['recall@10_final']:.3f} " + f"nDCG@10(fin)={bucket['ndcg@10_final']:.3f} " + f"hops={bucket['queries_with_crossref_hop']}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/smoke_e2e.py b/eval/smoke_e2e.py new file mode 100644 index 00000000..aa5ef625 --- /dev/null +++ b/eval/smoke_e2e.py @@ -0,0 +1,365 @@ +"""Scripted end-to-end smoke test (Phase 0.4). + +Starts the two Python services as child processes against a throwaway SQLite +database and a throwaway LanceDB directory, drives them over HTTP exactly the +way the browser does, and asserts on the answers. Nothing touches the developer's +real ``backend/chat_data.db`` or ``./lancedb``. + + 1. POST :8000/indexes create an index row + 2. POST :8000/indexes/<id>/upload upload the Atlas-7 planted-fact PDF + 3. POST :8000/indexes/<id>/build build it (delegates to :8001/index) + 4. POST :8000/sessions create a session + 5. POST :8000/sessions/<sid>/indexes/<iid> link them + 6. POST :8000/sessions/<sid>/messages 4 planted-fact questions, non-streaming + asserts: planted fact present in the answer, source_documents non-empty, + a [Confidence: N%] tag on the answer, message_count == 2 * turns + 7. POST :8000/sessions/<sid>/messages/save persist a streamed turn with steps+sources + 8. GET :8000/sessions/<sid> assert both round-trip out of SQLite + +Teardown (children killed, temp dirs removed) runs even when an assertion fails +or the run is interrupted. Exit code 0 = every assertion passed, 1 = at least one +failed or the services never came up. + +Pre-flight: if :8000 or :8001 is already accepting connections (e.g. the +developer's own stack), the run aborts before spawning anything โ€” otherwise the +health checks would 200 against that pre-existing service and the smoke run +would POST uploads/builds/messages into the REAL ``backend/chat_data.db`` and +``./lancedb``. + + .venv/bin/python eval/smoke_e2e.py +""" + +import argparse +import json +import os +import re +import shutil +import signal +import socket +import subprocess +import sys +import tempfile +import time + +import requests + +EVAL_DIR = os.path.dirname(os.path.abspath(__file__)) +REPO_ROOT = os.path.abspath(os.path.join(EVAL_DIR, "..")) +RUN_STARTED_AT = time.time() +TEST_PDF = os.path.join(EVAL_DIR, "corpora", "atlas7_service_manual.pdf") + +BACKEND = "http://localhost:8000" +RAG_API = "http://localhost:8001" + +EMBEDDER = os.getenv("EMBEDDING_MODEL", "Qwen/Qwen3-Embedding-0.6B") + +# (question, substring that must appear in the answer) +QUESTIONS = [ + ("What pressure does the brew boiler operate at during extraction?", "9.2"), + ("Which sensor part should be replaced when error code E11 appears?", "TS-71"), + ("How long is the Atlas-7 parts warranty?", "36"), + ("Where is the serial number engraved?", "drip tray"), +] + +CONFIDENCE_RE = re.compile(r"\[Confidence:\s*(\d+)%\]") + +# The gateway stores uploads under <repo>/shared_uploads/, outside the temp dir +# it knows nothing about, so teardown has to clean them up explicitly. +UPLOADED_PATHS: list = [] + + +class Results: + """Collects one pass/fail line per assertion so the report is complete.""" + + def __init__(self): + self.rows = [] + + def check(self, name: str, ok: bool, detail: str = "") -> bool: + self.rows.append((name, bool(ok), detail)) + print(f" [{'PASS' if ok else 'FAIL'}] {name}{(' โ€” ' + detail) if detail else ''}") + return bool(ok) + + @property + def failed(self): + return [r for r in self.rows if not r[1]] + + def report(self) -> int: + print("\n" + "=" * 72) + print(f"{len(self.rows) - len(self.failed)}/{len(self.rows)} assertions passed") + for name, ok, detail in self.rows: + if not ok: + print(f" FAILED: {name} โ€” {detail}") + print("=" * 72) + return 1 if self.failed else 0 + + +def port_accepting_connections(port: int, host: str = "localhost") -> bool: + """True when something is already listening on this port.""" + try: + with socket.create_connection((host, port), timeout=2): + return True + except OSError: + return False + + +def wait_for(url: str, timeout: float, proc: subprocess.Popen, log_path: str) -> bool: + deadline = time.time() + timeout + while time.time() < deadline: + if proc.poll() is not None: + print(f" process exited early with code {proc.returncode}; see {log_path}") + return False + try: + if requests.get(url, timeout=3).status_code == 200: + # A 200 while the child is dead means the port belongs to someone + # else's service โ€” never declare that healthy. + if proc.poll() is None: + return True + print(f" health 200 but the child exited with code " + f"{proc.returncode} โ€” something else owns this port") + return False + except requests.RequestException: + pass + time.sleep(1.0) + return False + + +def start_services(env: dict, log_dir: str, timeout: float): + """Start the RAG API then the gateway; return (procs, log_paths, ok).""" + procs, logs = [], {} + + for name, command, health in ( + ("rag-api", [sys.executable, "-m", "rag_system.api_server"], f"{RAG_API}/health"), + ("backend", [sys.executable, "backend/server.py"], f"{BACKEND}/health"), + ): + log_path = os.path.join(log_dir, f"{name}.log") + logs[name] = log_path + handle = open(log_path, "w", encoding="utf-8") + print(f" starting {name} โ†’ {log_path}") + proc = subprocess.Popen( + command, cwd=REPO_ROOT, env=env, + stdout=handle, stderr=subprocess.STDOUT, + start_new_session=True, # own process group, so teardown kills children too + ) + procs.append((name, proc, handle)) + if not wait_for(health, timeout, proc, log_path): + print(f" โŒ {name} did not become healthy within {timeout:.0f}s") + return procs, logs, False + print(f" {name} healthy") + + return procs, logs, True + + +def stop_services(procs) -> None: + for name, proc, handle in reversed(procs): + if proc.poll() is None: + try: + os.killpg(os.getpgid(proc.pid), signal.SIGTERM) + except (ProcessLookupError, PermissionError): + proc.terminate() + try: + proc.wait(timeout=15) + except subprocess.TimeoutExpired: + try: + os.killpg(os.getpgid(proc.pid), signal.SIGKILL) + except (ProcessLookupError, PermissionError): + proc.kill() + proc.wait(timeout=10) + print(f" stopped {name} (exit {proc.returncode})") + handle.close() + + +def run_smoke(results: Results, timeout: float) -> None: + print("\n--- 1..3 index the planted-fact PDF over HTTP") + resp = requests.post(f"{BACKEND}/indexes", + json={"name": "smoke-atlas7", "description": "Phase 0 smoke"}, + timeout=60) + resp.raise_for_status() + index_id = resp.json()["index_id"] + print(f" index_id {index_id}") + + with open(TEST_PDF, "rb") as fh: + upload = requests.post(f"{BACKEND}/indexes/{index_id}/upload", + files={"files": ("atlas7_service_manual.pdf", fh, "application/pdf")}, + timeout=120) + upload.raise_for_status() + uploaded = upload.json().get("uploaded_files", []) + UPLOADED_PATHS.extend(f["stored_path"] for f in uploaded if f.get("stored_path")) + results.check("upload accepted the PDF", len(uploaded) == 1, + json.dumps(upload.json())[:160]) + + build = requests.post(f"{BACKEND}/indexes/{index_id}/build", + json={"enable_enrich": False, "chunk_size": 512, + "embedding_model": EMBEDDER}, + timeout=timeout) + build_body = build.json() + results.check("index build returned 200 with no error", + build.status_code == 200 and "error" not in build_body, + f"status={build.status_code} body={json.dumps(build_body)[:220]}") + + print("\n--- 4..5 session + link") + session_id = requests.post(f"{BACKEND}/sessions", json={"title": "smoke"}, + timeout=30).json()["session_id"] + link = requests.post(f"{BACKEND}/sessions/{session_id}/indexes/{index_id}", timeout=30) + results.check("index linked to session", link.status_code == 200, link.text[:160]) + print(f" session_id {session_id}") + + print("\n--- 6 four planted-fact questions (non-streaming, force_rag)") + for turn, (question, expected) in enumerate(QUESTIONS, start=1): + label = f"q{turn}" + chat = requests.post(f"{BACKEND}/sessions/{session_id}/messages", + json={"message": question, "force_rag": True, "verify": True}, + timeout=timeout) + if chat.status_code != 200: + results.check(f"{label}: chat returned 200", False, + f"status={chat.status_code} body={chat.text[:200]}") + continue + body = chat.json() + answer = body.get("response", "") or "" + sources = body.get("source_documents") or [] + message_count = (body.get("session") or {}).get("message_count") + + results.check(f"{label}: planted fact '{expected}' in answer", + expected.lower() in answer.lower(), + f"{question!r} -> {answer[:200]!r}") + results.check(f"{label}: source_documents non-empty", + len(sources) > 0, f"{len(sources)} sources") + confidence = CONFIDENCE_RE.search(answer) + results.check(f"{label}: [Confidence: N%] tag present", + confidence is not None, + f"tag={confidence.group(0) if confidence else 'absent'}") + results.check(f"{label}: message_count == {2 * turn}", + message_count == 2 * turn, f"got {message_count}") + + print("\n--- 7..8 streamed-turn persistence round-trip") + saved_user = "What is the descaling interval?" + saved_answer = "Descaling must be performed every 60 days when water hardness exceeds 120 ppm." + saved_sources = [{ + "chunk_id": "smoke-chunk-1", + "text": "Descaling must be performed every 60 days when water hardness exceeds 120 ppm.", + "document_id": "atlas7_service_manual.pdf", + "chunk_index": 0, + "score": 0.42, + }] + saved_steps = [ + {"type": "retrieval_started", "data": {"mode": "hybrid"}}, + {"type": "rerank_done", "data": {"count": 10}}, + ] + save = requests.post(f"{BACKEND}/sessions/{session_id}/messages/save", + json={"user_message": saved_user, "assistant_message": saved_answer, + "source_documents": saved_sources, "steps": saved_steps}, + timeout=60) + results.check("messages/save returned 200", save.status_code == 200, save.text[:200]) + + fetched = requests.get(f"{BACKEND}/sessions/{session_id}", timeout=60).json() + messages = fetched.get("messages", []) + assistant = next((m for m in messages + if m.get("sender") == "assistant" and m.get("content") == saved_answer), None) + results.check("saved assistant message round-trips out of SQLite", + assistant is not None, f"{len(messages)} messages in session") + + metadata = (assistant or {}).get("metadata") or {} + round_tripped_sources = metadata.get("source_documents") or [] + results.check("saved source_documents round-trip in metadata", + len(round_tripped_sources) == 1 + and round_tripped_sources[0].get("chunk_id") == "smoke-chunk-1", + json.dumps(round_tripped_sources)[:200]) + round_tripped_steps = metadata.get("steps") or [] + results.check("saved steps round-trip in metadata", + [s.get("type") for s in round_tripped_steps] + == ["retrieval_started", "rerank_done"], + json.dumps(round_tripped_steps)[:200]) + + expected_total = 2 * len(QUESTIONS) + 2 + results.check(f"final message_count == {expected_total}", + (fetched.get("session") or {}).get("message_count") == expected_total, + f"got {(fetched.get('session') or {}).get('message_count')}") + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--timeout", type=float, default=900.0, + help="per-request / per-service-startup timeout in seconds") + parser.add_argument("--keep-temp", action="store_true", help="do not delete the temp dir") + args = parser.parse_args() + + if not os.path.exists(TEST_PDF): + print(f"missing test PDF: {TEST_PDF}") + return 1 + + # Pre-flight, BEFORE any child is spawned: if either port is already serving, + # the health checks would pass against that pre-existing service and the run + # would drive uploads/builds/messages into the developer's REAL + # backend/chat_data.db and ./lancedb. + for name, port in (("backend", 8000), ("rag-api", 8001)): + if port_accepting_connections(port): + print(f"โŒ port {port} is already accepting connections โ€” is your own " + f"{name} stack running? The smoke run would POST into IT, not " + f"into its throwaway children. Stop whatever listens on :{port} " + f"and re-run.") + return 1 + + temp_root = tempfile.mkdtemp(prefix="localgpt-smoke-") + log_dir = os.path.join(temp_root, "logs") + os.makedirs(log_dir, exist_ok=True) + + env = os.environ.copy() + env.update({ + "EMBEDDING_MODEL": EMBEDDER, + "DB_PATH": os.path.join(temp_root, "smoke_chat.db"), + "LANCEDB_PATH": os.path.join(temp_root, "lancedb"), + "RAG_API_URL": RAG_API, + "PYTHONUNBUFFERED": "1", + "TOKENIZERS_PARALLELISM": "false", + }) + + print("localGPT end-to-end smoke") + print(f" embedder {EMBEDDER}") + print(f" temp dir {temp_root}") + + results = Results() + procs = [] + started = time.time() + try: + procs, logs, ok = start_services(env, log_dir, args.timeout) + if not ok: + results.check("both services became healthy", False, + f"logs in {log_dir}") + else: + results.check("both services became healthy", True) + run_smoke(results, args.timeout) + except Exception as e: # noqa: BLE001 โ€” a crash here is a smoke failure, not a traceback + results.check("smoke run completed without raising", False, f"{type(e).__name__}: {e}") + finally: + print("\n--- teardown") + stop_services(procs) + for path in UPLOADED_PATHS: + try: + os.remove(path) + print(f" removed upload {path}") + except OSError as e: + print(f" could not remove upload {path}: {e}") + # The overview builder writes to <repo cwd>/index_store/overviews/ (not + # env-redirectable), so delete any overview files created during this run. + overview_dir = os.path.join(REPO_ROOT, "index_store", "overviews") + if os.path.isdir(overview_dir): + for name in os.listdir(overview_dir): + p = os.path.join(overview_dir, name) + try: + if os.path.getmtime(p) >= RUN_STARTED_AT: + os.remove(p) + print(f" removed leaked overview {p}") + except OSError: + pass + if args.keep_temp: + print(f" kept {temp_root}") + else: + shutil.rmtree(temp_root, ignore_errors=True) + print(f" removed {temp_root}") + print(f" wall clock {time.time() - started:.1f}s") + + return results.report() + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/validate_judge_hard.py b/eval/validate_judge_hard.py new file mode 100644 index 00000000..aef32628 --- /dev/null +++ b/eval/validate_judge_hard.py @@ -0,0 +1,116 @@ +"""Score a judge against eval/judge_hard_cases.jsonl. + +These are the 18 real system answers from the 2026-08-12 escalation re-run +whose ground truth ("does the answer contain the gold fact?") was hand- +adjudicated (eval/decisions/phase4-escalation-rerun.md ยง6). The qwen3.5:4b +judge's 5-vote majority is wrong on 5 of the 18 โ€” all five being answers that +contain the gold fact verbatim but were voted down. Any candidate judge must +beat 13/18 here to be worth switching to. + +Slot assignment matches the A/B harness: EVIDENCE = the system's answer, +ANSWER = the gold answer, so grounded=true means "the system's answer contains +the gold fact". + +Voting honesty: ``--votes`` must be a positive odd number (an even count can +tie, and a silent tie-break is not a verdict). A judge call whose verdict fails +to parse (``grounded`` is None) is recorded as an ERROR, never coerced to a +vote โ€” ``bool(None)`` would bank free "correct" votes on every label-False +row. A row whose valid votes tie (possible once errors shrink the odd total), +or that has no valid votes at all, is an error row: reported as such and +counted neither correct nor incorrect โ€” the same way ``judge.py --validate`` +keeps unparseable verdicts out of its confusion matrix. + +Usage:: + + .venv/bin/python eval/validate_judge_hard.py # local default + JUDGE_MODEL=claude-sonnet-5 .venv/bin/python eval/validate_judge_hard.py + .venv/bin/python eval/validate_judge_hard.py --model claude-sonnet-5 --votes 3 +""" + +import argparse +import json +import os +import sys +from datetime import datetime, timezone + +EVAL_DIR = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, EVAL_DIR) +sys.path.insert(0, os.path.abspath(os.path.join(EVAL_DIR, ".."))) + +from judge import DEFAULT_MODEL, DEFAULT_VERSION, GroundednessJudge # noqa: E402 + +CASES_PATH = os.path.join(EVAL_DIR, "judge_hard_cases.jsonl") +RESULTS_DIR = os.path.join(EVAL_DIR, "results") + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--prompt-version", default=DEFAULT_VERSION) + parser.add_argument("--votes", type=int, default=1, + help="votes per row; majority decides (must be odd)") + args = parser.parse_args() + if args.votes < 1 or args.votes % 2 == 0: + parser.error("--votes must be a positive odd number โ€” an even count can " + "tie, and a tie is not a verdict") + + with open(CASES_PATH, "r", encoding="utf-8") as fh: + cases = [json.loads(line) for line in fh if line.strip()] + + judge = GroundednessJudge(model=args.model, version=args.prompt_version) + correct = 0 + error_rows = 0 + qwen_correct = 0 + rows = [] + for case in cases: + votes = [] + errors = 0 + reasons = [] + for _ in range(args.votes): + verdict = judge.judge(case["question"], case["gold_answer"], case["answer"]) + # A parse/shape failure is an error, never a vote: coercing + # grounded=None to False would bank free "correct" votes on every + # label-False row. + if verdict.get("error") or verdict["grounded"] is None: + errors += 1 + else: + votes.append(verdict["grounded"]) + reasons.append(verdict.get("reason", "")) + # Majority over the valid votes only. A tie among them (possible once + # errors shrink the odd total) is undecided โ€” reported, not forced. + predicted = None if not votes or 2 * sum(votes) == len(votes) \ + else sum(votes) > len(votes) / 2 + label = case["label_grounded"] + ok = None if predicted is None else predicted == label + correct += (ok is True) + error_rows += (predicted is None) + qwen_ok = (case["qwen4b_votes"] >= 3) == label + qwen_correct += qwen_ok + rows.append({**{k: case[k] for k in ("id", "label_grounded", "qwen4b_votes")}, + "votes": votes, "error_votes": errors, "predicted": predicted, + "correct": ok, "reasons": reasons}) + flag = " " if ok else ("!" if ok is None else "<") + print(f" {flag} {case['id']:<22} label={'contains-fact' if label else 'missing-fact'}" + f" pred={str(predicted):<5} votes={sum(votes)}/{len(votes)}" + f"{f' +{errors} error(s)' if errors else ''}" + f" (4b was {case['qwen4b_votes']}/5{'' if qwen_ok else ' WRONG'})") + + n = len(cases) + print(f"\n {args.model} (k={args.votes}): {correct}/{n} correct" + + (f" (+{error_rows} error row(s), counted neither correct nor incorrect)" + if error_rows else "")) + print(f" qwen3.5:4b baseline (k=5): {qwen_correct}/{n} correct") + + os.makedirs(RESULTS_DIR, exist_ok=True) + stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + safe_model = args.model.replace(":", "_").replace("/", "_") + out_path = os.path.join(RESULTS_DIR, f"judge_hard_{safe_model}_{stamp}.json") + with open(out_path, "w", encoding="utf-8") as fh: + json.dump({"model": args.model, "votes": args.votes, "correct": correct, + "error_rows": error_rows, "n": n, "rows": rows}, fh, indent=2) + print(f" written {out_path}") + return 0 if correct > qwen_correct else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/verify_crossref_goldset.py b/eval/verify_crossref_goldset.py new file mode 100644 index 00000000..9cf0d4a1 --- /dev/null +++ b/eval/verify_crossref_goldset.py @@ -0,0 +1,138 @@ +"""Row-level verification for the hand-authored `acquisition` gold set. + +`eval/corpora/verify_facts.py` is gate 1 for the *facts* sidecar and +`run_eval.py --coverage-only` is gate 2 for reachability after chunking. Neither +checks the properties that make the cross-reference rows mean anything, so this +does: + + 1. every ``expected`` string occurs in the document named in ``expected_sources`` + 2. the query does not contain any of its ``expected`` strings verbatim + (whitespace-normalised, case-insensitive) โ€” no answer leak + 3. every ``expected`` string occurs in **exactly one** of the ten documents, so + "the answer lives in a different document" is a checkable claim + 4. ``fact_ids`` resolve to ``acquisition.facts.json`` with the same text and source + 5. ``requires_crossref`` is exactly + ``anchor_doc is not None and any(source != anchor_doc)`` + 6. ``multi_document`` is exactly ``len(set(expected_sources)) > 1`` + 7. every ``anchor_doc`` names one of the ten documents (``expected_sources`` + entries already fail check 1 when they name no document) + + .venv/bin/python eval/verify_crossref_goldset.py + +The tallies it prints are the ones quoted in BASELINE.md ยง "Phase 4 baseline". +""" + +import json +import os +import sys +from collections import Counter + +EVAL_DIR = os.path.dirname(os.path.abspath(__file__)) +CORPUS_DIR = os.path.join(EVAL_DIR, "corpora", "acquisition") +SIDECAR = os.path.join(EVAL_DIR, "corpora", "acquisition.facts.json") +GOLD = os.path.join(EVAL_DIR, "goldset", "acquisition.jsonl") + + +def norm(text: str) -> str: + return " ".join((text or "").split()).lower() + + +def document_texts() -> dict: + import pymupdf + + texts = {} + for name in sorted(os.listdir(CORPUS_DIR)): + if not name.endswith(".pdf"): + continue + doc = pymupdf.open(os.path.join(CORPUS_DIR, name)) + try: + texts[name] = norm(" ".join(page.get_text() for page in doc)) + finally: + doc.close() + return texts + + +def main() -> int: + texts = document_texts() + with open(SIDECAR, "r", encoding="utf-8") as fh: + facts = {f["id"]: f for f in json.load(fh)["facts"]} + with open(GOLD, "r", encoding="utf-8") as fh: + rows = [json.loads(line) for line in fh if line.strip()] + + problems = [] + tally = Counter() + n_expected = 0 + + for row in rows: + sources = row["expected_sources"] + if len(sources) != len(row["expected"]) or len(sources) != len(row["fact_ids"]): + problems.append(f"{row['id']}: expected / expected_sources / fact_ids length mismatch") + continue + + for text, source, fact_id in zip(row["expected"], sources, row["fact_ids"]): + n_expected += 1 + + if source in texts and norm(text) in texts[source]: + tally["expected_in_source"] += 1 + else: + problems.append(f"{row['id']}: {text!r} not found in {source}") + + if norm(text) not in norm(row["query"]): + tally["no_verbatim_leak"] += 1 + else: + problems.append(f"{row['id']}: query leaks the expected string {text!r}") + + holders = [d for d, t in texts.items() if norm(text) in t] + if holders == [source]: + tally["unique_to_source"] += 1 + else: + problems.append(f"{row['id']}: {text!r} occurs in {holders}, not only {source}") + + fact = facts.get(fact_id) + if fact and norm(fact["expected"]) == norm(text) and fact["source"] == source: + tally["fact_id_resolves"] += 1 + else: + problems.append(f"{row['id']}: fact id {fact_id!r} does not match the sidecar") + + anchor = row["anchor_doc"] + if anchor is not None and anchor not in texts: + problems.append(f"{row['id']}: anchor_doc {anchor!r} is not a corpus document") + expected_crossref = anchor is not None and any(s != anchor for s in sources) + if row["dimensions"]["requires_crossref"] is expected_crossref: + tally["crossref_flag_consistent"] += 1 + else: + problems.append(f"{row['id']}: requires_crossref should be {expected_crossref}") + + if row["multi_document"] is (len(set(sources)) > 1): + tally["multi_document_consistent"] += 1 + else: + problems.append(f"{row['id']}: multi_document flag is wrong") + + print(f"{GOLD}: {len(rows)} rows, {n_expected} expected strings, " + f"{len(texts)} source documents") + for key in ("expected_in_source", "no_verbatim_leak", "unique_to_source", + "fact_id_resolves"): + print(f" {key:<26} {tally[key]}/{n_expected}") + for key in ("crossref_flag_consistent", "multi_document_consistent"): + print(f" {key:<26} {tally[key]}/{len(rows)}") + + crossref = [r for r in rows if r["dimensions"]["requires_crossref"]] + multi = [r for r in rows if r["multi_document"]] + print(f" requires_crossref=true {len(crossref)}") + print(f" multi_document=true {len(multi)}") + print(f" question_type " + f"{dict(Counter(r['dimensions']['question_type'] for r in rows))}") + print(f" difficulty " + f"{dict(Counter(r['dimensions']['difficulty'] for r in rows))}") + + if problems: + print(f"\n{len(problems)} PROBLEM(S):") + for problem in problems: + print(" " + problem) + return 1 + print("\nall row-level checks passed.") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/verify_rfc_goldset.py b/eval/verify_rfc_goldset.py new file mode 100644 index 00000000..9a41b7c4 --- /dev/null +++ b/eval/verify_rfc_goldset.py @@ -0,0 +1,163 @@ +"""Row-level verification for the hand-authored `rfc` gold set. + +The same gate `eval/verify_crossref_goldset.py` applies to `acquisition`, ported +to a corpus of 23 real IETF RFCs. It checks the properties that make the +cross-reference rows mean anything, on the actual downloaded text: + + 1. every ``expected`` string occurs in the document named in ``expected_sources`` + 2. the query does not contain any of its ``expected`` strings verbatim + (whitespace-normalised, case-insensitive) โ€” no answer leak + 3. every ``expected`` string occurs in **exactly one** of the 23 documents, so + "the answer lives in a different document" is a checkable claim. + ``UNIQUENESS_EXEMPT`` below records the strings where that is impossible in + an RFC corpus, with the reason; those are counted separately, never silently. + 4. ``fact_ids`` resolve to ``eval/corpora/rfc/rfc.facts.json`` with the same + text and source + 5. ``requires_crossref`` is exactly + ``anchor_doc is not None and any(source != anchor_doc)`` + 6. ``multi_document`` is exactly ``len(set(expected_sources)) > 1`` + 7. every ``anchor_doc`` and ``expected_sources`` entry names a file that exists + +Comparison is on whitespace-normalised, lowercased text, because RFCs are +hard-wrapped at 72 columns: a sentence-length anchor necessarily spans a line +break. ``eval/run_eval.py`` scores chunk relevance with exactly the same +normalisation (``run_eval.norm``), so a string that passes here is a string the +metric can match. + + .venv/bin/python eval/verify_rfc_goldset.py +""" + +import json +import os +import sys +from collections import Counter + +EVAL_DIR = os.path.dirname(os.path.abspath(__file__)) +CORPUS_DIR = os.path.join(EVAL_DIR, "corpora", "rfc") +SIDECAR = os.path.join(CORPUS_DIR, "rfc.facts.json") +GOLD = os.path.join(EVAL_DIR, "goldset", "rfc.jsonl") + +# Strings that cannot be unique to one document in this corpus, with the reason. +# Nothing is exempted for convenience: each entry is a phrase the RFC series +# repeats by construction. An exempt string is still required to be present in +# its named source (check 1) โ€” only check 3 is relaxed, and the count is +# reported separately so the tally can never hide behind a total. +UNIQUENESS_EXEMPT = { + # (none needed today โ€” kept as the documented mechanism, since RFC + # boilerplate such as the BCP 14 sentence appears in all 23 documents and any + # future row anchored on one would have to be recorded here.) +} + + +def norm(text: str) -> str: + return " ".join((text or "").split()).lower() + + +def document_texts() -> dict: + texts = {} + for name in sorted(os.listdir(CORPUS_DIR)): + if not name.endswith(".txt"): + continue + with open(os.path.join(CORPUS_DIR, name), "r", encoding="utf-8", + errors="replace") as fh: + texts[name] = norm(fh.read()) + return texts + + +def main() -> int: + texts = document_texts() + with open(SIDECAR, "r", encoding="utf-8") as fh: + facts = {f["id"]: f for f in json.load(fh)["facts"]} + with open(GOLD, "r", encoding="utf-8") as fh: + rows = [json.loads(line) for line in fh if line.strip()] + + problems = [] + tally = Counter() + n_expected = 0 + + for row in rows: + sources = row["expected_sources"] + if len(sources) != len(row["expected"]) or len(sources) != len(row["fact_ids"]): + problems.append(f"{row['id']}: expected / expected_sources / fact_ids length mismatch") + continue + + for text, source, fact_id in zip(row["expected"], sources, row["fact_ids"]): + n_expected += 1 + + if source in texts and norm(text) in texts[source]: + tally["expected_in_source"] += 1 + else: + problems.append(f"{row['id']}: {text!r} not found in {source}") + + if norm(text) not in norm(row["query"]): + tally["no_verbatim_leak"] += 1 + else: + problems.append(f"{row['id']}: query leaks the expected string {text!r}") + + holders = [d for d, t in texts.items() if norm(text) in t] + if holders == [source]: + tally["unique_to_source"] += 1 + elif text in UNIQUENESS_EXEMPT: + tally["uniqueness_exempt"] += 1 + else: + problems.append( + f"{row['id']}: {text!r} occurs in {holders}, not only {source} " + f"(add it to UNIQUENESS_EXEMPT with a reason if that is unavoidable)") + + fact = facts.get(fact_id) + if fact and norm(fact["expected"]) == norm(text) and fact["source"] == source: + tally["fact_id_resolves"] += 1 + else: + problems.append(f"{row['id']}: fact id {fact_id!r} does not match the sidecar") + + anchor = row["anchor_doc"] + if anchor is not None and anchor not in texts: + problems.append(f"{row['id']}: anchor_doc {anchor!r} is not a corpus document") + expected_crossref = anchor is not None and any(s != anchor for s in sources) + if row["dimensions"]["requires_crossref"] is expected_crossref: + tally["crossref_flag_consistent"] += 1 + else: + problems.append(f"{row['id']}: requires_crossref should be {expected_crossref}") + + if row["multi_document"] is (len(set(sources)) > 1): + tally["multi_document_consistent"] += 1 + else: + problems.append(f"{row['id']}: multi_document flag is wrong") + + unused = sorted(set(facts) - {f for r in rows for f in r["fact_ids"]}) + + print(f"{GOLD}: {len(rows)} rows, {n_expected} expected strings, " + f"{len(texts)} source documents") + for key in ("expected_in_source", "no_verbatim_leak", "unique_to_source", + "fact_id_resolves"): + print(f" {key:<26} {tally[key]}/{n_expected}") + if tally["uniqueness_exempt"]: + print(f" {'uniqueness_exempt':<26} {tally['uniqueness_exempt']} " + f"(recorded in UNIQUENESS_EXEMPT, not counted as unique)") + for key in ("crossref_flag_consistent", "multi_document_consistent"): + print(f" {key:<26} {tally[key]}/{len(rows)}") + + crossref = [r for r in rows if r["dimensions"]["requires_crossref"]] + multi = [r for r in rows if r["multi_document"]] + print(f" requires_crossref=true {len(crossref)}") + print(f" multi_document=true {len(multi)}") + print(f" question_type " + f"{dict(Counter(r['dimensions']['question_type'] for r in rows))}") + print(f" difficulty " + f"{dict(Counter(r['dimensions']['difficulty'] for r in rows))}") + print(f" documents referenced " + f"{len({s for r in rows for s in r['expected_sources']})}/{len(texts)}") + if unused: + print(f" sidecar facts unused by any row: {len(unused)}") + + if problems: + print(f"\n{len(problems)} PROBLEM(S):") + for problem in problems: + print(" " + problem) + return 1 + print("\nall row-level checks passed.") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/next.config.ts b/next.config.ts index 3d307dfa..b0e4536c 100644 --- a/next.config.ts +++ b/next.config.ts @@ -4,11 +4,44 @@ const nextConfig: NextConfig = { /* config options here */ eslint: { // Warning: This allows production builds to successfully complete even if your project has ESLint errors. + // NOTE: `next lint` is deprecated in Next 15 and the repo has pre-existing + // ESLint errors (mostly no-explicit-any) โ€” this gate stays on deliberately. ignoreDuringBuilds: true, }, typescript: { - // Warning: This allows production builds to successfully complete even if your project has type errors. - ignoreBuildErrors: true, + // Type-checked: `npx tsc --noEmit` is clean, so let build failures surface. + ignoreBuildErrors: false, + }, + webpack: (config, { dev }) => { + if (dev) { + // The frontend shares its root with the Python backend. The gateway + // writes backend/chat_data.db on every persisted chat turn; if the dev + // watcher sees those writes it Fast-Refreshes mid-stream, remounting the + // page and aborting the in-flight SSE fetch (net::ERR_ABORTED) โ€” which + // presents as "streaming stopped working". Keep the watcher on frontend + // sources only. + config.watchOptions = { + ...config.watchOptions, + ignored: [ + "**/node_modules/**", + "**/.git/**", + "**/.next/**", + "**/backend/**", + "**/rag_system/**", + "**/eval/**", + "**/lancedb/**", + "**/index_store/**", + "**/shared_uploads/**", + "**/logs/**", + "**/.venv/**", + "**/Documentation/**", + "**/*.db", + "**/*.db-wal", + "**/*.db-shm", + ], + }; + } + return config; }, }; diff --git a/package-lock.json b/package-lock.json index 7eb5b4df..10907fb2 100644 --- a/package-lock.json +++ b/package-lock.json @@ -9,18 +9,16 @@ "version": "0.1.0", "dependencies": { "@radix-ui/react-avatar": "^1.1.10", - "@radix-ui/react-dropdown-menu": "^2.1.15", "@radix-ui/react-scroll-area": "^1.2.9", - "@radix-ui/react-separator": "^1.1.7", "@radix-ui/react-slot": "^1.2.3", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", - "framer-motion": "^12.16.0", "lucide-react": "^0.513.0", "next": "15.3.3", "react": "^19.0.0", "react-dom": "^19.0.0", "react-markdown": "^10.1.0", + "remark-breaks": "^4.0.0", "remark-gfm": "^4.0.1", "tailwind-merge": "^3.3.0" }, @@ -238,44 +236,6 @@ "node": "^18.18.0 || ^20.9.0 || >=21.1.0" } }, - "node_modules/@floating-ui/core": { - "version": "1.7.1", - "resolved": "https://registry.npmjs.org/@floating-ui/core/-/core-1.7.1.tgz", - "integrity": "sha512-azI0DrjMMfIug/ExbBaeDVJXcY0a7EPvPjb2xAJPa4HeimBX+Z18HK8QQR3jb6356SnDDdxx+hinMLcJEDdOjw==", - "license": "MIT", - "dependencies": { - "@floating-ui/utils": "^0.2.9" - } - }, - "node_modules/@floating-ui/dom": { - "version": "1.7.1", - "resolved": "https://registry.npmjs.org/@floating-ui/dom/-/dom-1.7.1.tgz", - "integrity": "sha512-cwsmW/zyw5ltYTUeeYJ60CnQuPqmGwuGVhG9w0PRaRKkAyi38BT5CKrpIbb+jtahSwUl04cWzSx9ZOIxeS6RsQ==", - "license": "MIT", - "dependencies": { - "@floating-ui/core": "^1.7.1", - "@floating-ui/utils": "^0.2.9" - } - }, - "node_modules/@floating-ui/react-dom": { - "version": "2.1.3", - "resolved": "https://registry.npmjs.org/@floating-ui/react-dom/-/react-dom-2.1.3.tgz", - "integrity": "sha512-huMBfiU9UnQ2oBwIhgzyIiSpVgvlDstU8CX0AF+wS+KzmYMs0J2a3GwuFHV1Lz+jlrQGeC1fF+Nv0QoumyV0bA==", - "license": "MIT", - "dependencies": { - "@floating-ui/dom": "^1.0.0" - }, - "peerDependencies": { - "react": ">=16.8.0", - "react-dom": ">=16.8.0" - } - }, - "node_modules/@floating-ui/utils": { - "version": "0.2.9", - "resolved": "https://registry.npmjs.org/@floating-ui/utils/-/utils-0.2.9.tgz", - "integrity": "sha512-MDWhGtE+eHw5JW7lq4qhc5yRLS11ERl1c7Z6Xd0a58DozHES6EnNNwUWbMiG4J9Cgj053Bhk8zvlhFYKVhULwg==", - "license": "MIT" - }, "node_modules/@humanfs/core": { "version": "0.19.1", "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.1.tgz", @@ -1021,29 +981,6 @@ "integrity": "sha512-XnbHrrprsNqZKQhStrSwgRUQzoCI1glLzdw79xiZPoofhGICeZRSQ3dIxAKH1gb3OHfNf4d6f+vAv3kil2eggA==", "license": "MIT" }, - "node_modules/@radix-ui/react-arrow": { - "version": "1.1.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-arrow/-/react-arrow-1.1.7.tgz", - "integrity": "sha512-F+M1tLhO+mlQaOWspE8Wstg+z6PwxwRd8oQ8IXceWz92kfAmalTRf0EjrouQeo7QssEPfCn05B4Ihs1K9WQ/7w==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-primitive": "2.1.3" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-avatar": { "version": "1.1.10", "resolved": "https://registry.npmjs.org/@radix-ui/react-avatar/-/react-avatar-1.1.10.tgz", @@ -1071,32 +1008,6 @@ } } }, - "node_modules/@radix-ui/react-collection": { - "version": "1.1.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-collection/-/react-collection-1.1.7.tgz", - "integrity": "sha512-Fh9rGN0MoI4ZFUNyfFVNU4y9LUz93u9/0K+yLgA2bwRojxM8JU1DyvvMBabnZPBgMWREAJvU2jjVzq+LrFUglw==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-slot": "1.2.3" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-compose-refs": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/@radix-ui/react-compose-refs/-/react-compose-refs-1.1.2.tgz", @@ -1142,216 +1053,6 @@ } } }, - "node_modules/@radix-ui/react-dismissable-layer": { - "version": "1.1.10", - "resolved": "https://registry.npmjs.org/@radix-ui/react-dismissable-layer/-/react-dismissable-layer-1.1.10.tgz", - "integrity": "sha512-IM1zzRV4W3HtVgftdQiiOmA0AdJlCtMLe00FXaHwgt3rAnNsIyDqshvkIW3hj/iu5hu8ERP7KIYki6NkqDxAwQ==", - "license": "MIT", - "dependencies": { - "@radix-ui/primitive": "1.1.2", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-callback-ref": "1.1.1", - "@radix-ui/react-use-escape-keydown": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-dropdown-menu": { - "version": "2.1.15", - "resolved": "https://registry.npmjs.org/@radix-ui/react-dropdown-menu/-/react-dropdown-menu-2.1.15.tgz", - "integrity": "sha512-mIBnOjgwo9AH3FyKaSWoSu/dYj6VdhJ7frEPiGTeXCdUFHjl9h3mFh2wwhEtINOmYXWhdpf1rY2minFsmaNgVQ==", - "license": "MIT", - "dependencies": { - "@radix-ui/primitive": "1.1.2", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-id": "1.1.1", - "@radix-ui/react-menu": "2.1.15", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-controllable-state": "1.2.2" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-focus-guards": { - "version": "1.1.2", - "resolved": "https://registry.npmjs.org/@radix-ui/react-focus-guards/-/react-focus-guards-1.1.2.tgz", - "integrity": "sha512-fyjAACV62oPV925xFCrH8DR5xWhg9KYtJT4s3u54jxp+L/hbpTY2kIeEFFbFe+a/HCE94zGQMZLIpVTPVZDhaA==", - "license": "MIT", - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-focus-scope": { - "version": "1.1.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-focus-scope/-/react-focus-scope-1.1.7.tgz", - "integrity": "sha512-t2ODlkXBQyn7jkl6TNaw/MtVEVvIGelJDCG41Okq/KwUsJBwQ4XVZsHAVUkK4mBv3ewiAS3PGuUWuY2BoK4ZUw==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-callback-ref": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-id": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/react-id/-/react-id-1.1.1.tgz", - "integrity": "sha512-kGkGegYIdQsOb4XjsfM97rXsiHaBwco+hFI66oO4s9LU+PLAC5oJ7khdOVFxkhsmlbpUqDAvXw11CluXP+jkHg==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-menu": { - "version": "2.1.15", - "resolved": "https://registry.npmjs.org/@radix-ui/react-menu/-/react-menu-2.1.15.tgz", - "integrity": "sha512-tVlmA3Vb9n8SZSd+YSbuFR66l87Wiy4du+YE+0hzKQEANA+7cWKH1WgqcEX4pXqxUFQKrWQGHdvEfw00TjFiew==", - "license": "MIT", - "dependencies": { - "@radix-ui/primitive": "1.1.2", - "@radix-ui/react-collection": "1.1.7", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-direction": "1.1.1", - "@radix-ui/react-dismissable-layer": "1.1.10", - "@radix-ui/react-focus-guards": "1.1.2", - "@radix-ui/react-focus-scope": "1.1.7", - "@radix-ui/react-id": "1.1.1", - "@radix-ui/react-popper": "1.2.7", - "@radix-ui/react-portal": "1.1.9", - "@radix-ui/react-presence": "1.1.4", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-roving-focus": "1.1.10", - "@radix-ui/react-slot": "1.2.3", - "@radix-ui/react-use-callback-ref": "1.1.1", - "aria-hidden": "^1.2.4", - "react-remove-scroll": "^2.6.3" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-popper": { - "version": "1.2.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-popper/-/react-popper-1.2.7.tgz", - "integrity": "sha512-IUFAccz1JyKcf/RjB552PlWwxjeCJB8/4KxT7EhBHOJM+mN7LdW+B3kacJXILm32xawcMMjb2i0cIZpo+f9kiQ==", - "license": "MIT", - "dependencies": { - "@floating-ui/react-dom": "^2.0.0", - "@radix-ui/react-arrow": "1.1.7", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-callback-ref": "1.1.1", - "@radix-ui/react-use-layout-effect": "1.1.1", - "@radix-ui/react-use-rect": "1.1.1", - "@radix-ui/react-use-size": "1.1.1", - "@radix-ui/rect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-portal": { - "version": "1.1.9", - "resolved": "https://registry.npmjs.org/@radix-ui/react-portal/-/react-portal-1.1.9.tgz", - "integrity": "sha512-bpIxvq03if6UNwXZ+HTK71JLh4APvnXntDc6XOX8UVq4XQOVl7lwok0AvIl+b8zgCw3fSaVTZMpAPPagXbKmHQ==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-presence": { "version": "1.1.4", "resolved": "https://registry.npmjs.org/@radix-ui/react-presence/-/react-presence-1.1.4.tgz", @@ -1399,37 +1100,6 @@ } } }, - "node_modules/@radix-ui/react-roving-focus": { - "version": "1.1.10", - "resolved": "https://registry.npmjs.org/@radix-ui/react-roving-focus/-/react-roving-focus-1.1.10.tgz", - "integrity": "sha512-dT9aOXUen9JSsxnMPv/0VqySQf5eDQ6LCk5Sw28kamz8wSOW2bJdlX2Bg5VUIIcV+6XlHpWTIuTPCf/UNIyq8Q==", - "license": "MIT", - "dependencies": { - "@radix-ui/primitive": "1.1.2", - "@radix-ui/react-collection": "1.1.7", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-direction": "1.1.1", - "@radix-ui/react-id": "1.1.1", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-callback-ref": "1.1.1", - "@radix-ui/react-use-controllable-state": "1.2.2" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-scroll-area": { "version": "1.2.9", "resolved": "https://registry.npmjs.org/@radix-ui/react-scroll-area/-/react-scroll-area-1.2.9.tgz", @@ -1461,29 +1131,6 @@ } } }, - "node_modules/@radix-ui/react-separator": { - "version": "1.1.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-separator/-/react-separator-1.1.7.tgz", - "integrity": "sha512-0HEb8R9E8A+jZjvmFCy/J4xhbXy3TV+9XSnGJ3KvTtjlIUy/YQ/p6UYZvi7YbeoeXdyU9+Y3scizK6hkY37baA==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-primitive": "2.1.3" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-slot": { "version": "1.2.3", "resolved": "https://registry.npmjs.org/@radix-ui/react-slot/-/react-slot-1.2.3.tgz", @@ -1517,61 +1164,6 @@ } } }, - "node_modules/@radix-ui/react-use-controllable-state": { - "version": "1.2.2", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-controllable-state/-/react-use-controllable-state-1.2.2.tgz", - "integrity": "sha512-BjasUjixPFdS+NKkypcyyN5Pmg83Olst0+c6vGov0diwTEo6mgdqVR6hxcEgFuh4QrAs7Rc+9KuGJ9TVCj0Zzg==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-effect-event": "0.0.2", - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-use-effect-event": { - "version": "0.0.2", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-effect-event/-/react-use-effect-event-0.0.2.tgz", - "integrity": "sha512-Qp8WbZOBe+blgpuUT+lw2xheLP8q0oatc9UpmiemEICxGvFLYmHm9QowVZGHtJlGbS6A6yJ3iViad/2cVjnOiA==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-use-escape-keydown": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-escape-keydown/-/react-use-escape-keydown-1.1.1.tgz", - "integrity": "sha512-Il0+boE7w/XebUHyBjroE+DbByORGR9KKmITzbR7MyQ4akpORYP/ZmbhAr0DG7RmmBqoOnZdy2QlvajJ2QA59g==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-callback-ref": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-use-is-hydrated": { "version": "0.1.0", "resolved": "https://registry.npmjs.org/@radix-ui/react-use-is-hydrated/-/react-use-is-hydrated-0.1.0.tgz", @@ -1605,48 +1197,6 @@ } } }, - "node_modules/@radix-ui/react-use-rect": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-rect/-/react-use-rect-1.1.1.tgz", - "integrity": "sha512-QTYuDesS0VtuHNNvMh+CjlKJ4LJickCMUAqjlE3+j8w+RlRpwyX3apEQKGFzbZGdo7XNG1tXa+bQqIE7HIXT2w==", - "license": "MIT", - "dependencies": { - "@radix-ui/rect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-use-size": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-size/-/react-use-size-1.1.1.tgz", - "integrity": "sha512-ewrXRDTAqAXlkl6t/fkXWNAhFX9I+CkKlw6zjEwk86RSPKwZr3xpBRso655aqYafwtnbpHLj6toFzmd6xdVptQ==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/rect": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/rect/-/rect-1.1.1.tgz", - "integrity": "sha512-HPwpGIzkl28mWyZqG52jiqDJ12waP11Pa1lGoiyUkIEuMLBP0oeK/C89esbXrxsky5we7dfd8U58nm0SgAWpVw==", - "license": "MIT" - }, "node_modules/@rtsao/scc": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/@rtsao/scc/-/scc-1.1.0.tgz", @@ -2657,18 +2207,6 @@ "dev": true, "license": "Python-2.0" }, - "node_modules/aria-hidden": { - "version": "1.2.6", - "resolved": "https://registry.npmjs.org/aria-hidden/-/aria-hidden-1.2.6.tgz", - "integrity": "sha512-ik3ZgC9dY/lYVVM++OISsaYDeg1tb0VtP5uL3ouh1koGOaUMDPpbFIei4JkFimWUFPn90sbMNMXQAIVOlnYKJA==", - "license": "MIT", - "dependencies": { - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - } - }, "node_modules/aria-query": { "version": "5.3.2", "resolved": "https://registry.npmjs.org/aria-query/-/aria-query-5.3.2.tgz", @@ -3364,12 +2902,6 @@ "node": ">=8" } }, - "node_modules/detect-node-es": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/detect-node-es/-/detect-node-es-1.1.0.tgz", - "integrity": "sha512-ypdmJU/TbBby2Dxibuv7ZLW3Bs1QEmM7nHjEANfohJLvE0XVujisn1qPJcZxg+qDucsr+bP6fLD1rPS3AhJ7EQ==", - "license": "MIT" - }, "node_modules/devlop": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/devlop/-/devlop-1.1.0.tgz", @@ -4205,33 +3737,6 @@ "url": "https://github.com/sponsors/ljharb" } }, - "node_modules/framer-motion": { - "version": "12.16.0", - "resolved": "https://registry.npmjs.org/framer-motion/-/framer-motion-12.16.0.tgz", - "integrity": "sha512-xryrmD4jSBQrS2IkMdcTmiS4aSKckbS7kLDCuhUn9110SQKG1w3zlq1RTqCblewg+ZYe+m3sdtzQA6cRwo5g8Q==", - "license": "MIT", - "dependencies": { - "motion-dom": "^12.16.0", - "motion-utils": "^12.12.1", - "tslib": "^2.4.0" - }, - "peerDependencies": { - "@emotion/is-prop-valid": "*", - "react": "^18.0.0 || ^19.0.0", - "react-dom": "^18.0.0 || ^19.0.0" - }, - "peerDependenciesMeta": { - "@emotion/is-prop-valid": { - "optional": true - }, - "react": { - "optional": true - }, - "react-dom": { - "optional": true - } - } - }, "node_modules/function-bind": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/function-bind/-/function-bind-1.1.2.tgz", @@ -4298,15 +3803,6 @@ "url": "https://github.com/sponsors/ljharb" } }, - "node_modules/get-nonce": { - "version": "1.0.1", - "resolved": "https://registry.npmjs.org/get-nonce/-/get-nonce-1.0.1.tgz", - "integrity": "sha512-FJhYRoDaiatfEkUK8HKlicmu/3SGFD51q3itKDGoSTysQJBnfOcxU5GxnhE1E6soB76MbT0MBtnKJuXyAx+96Q==", - "license": "MIT", - "engines": { - "node": ">=6" - } - }, "node_modules/get-proto": { "version": "1.0.1", "resolved": "https://registry.npmjs.org/get-proto/-/get-proto-1.0.1.tgz", @@ -5781,6 +5277,20 @@ "url": "https://opencollective.com/unified" } }, + "node_modules/mdast-util-newline-to-break": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/mdast-util-newline-to-break/-/mdast-util-newline-to-break-2.0.0.tgz", + "integrity": "sha512-MbgeFca0hLYIEx/2zGsszCSEJJ1JSCdiY5xQxRcLDDGa8EPvlLPupJ4DSajbMPAnC0je8jfb9TiUATnxxrHUog==", + "license": "MIT", + "dependencies": { + "@types/mdast": "^4.0.0", + "mdast-util-find-and-replace": "^3.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/unified" + } + }, "node_modules/mdast-util-phrasing": { "version": "4.1.0", "resolved": "https://registry.npmjs.org/mdast-util-phrasing/-/mdast-util-phrasing-4.1.0.tgz", @@ -6499,21 +6009,6 @@ "url": "https://github.com/sponsors/isaacs" } }, - "node_modules/motion-dom": { - "version": "12.16.0", - "resolved": "https://registry.npmjs.org/motion-dom/-/motion-dom-12.16.0.tgz", - "integrity": "sha512-Z2nGwWrrdH4egLEtgYMCEN4V2qQt1qxlKy/uV7w691ztyA41Q5Rbn0KNGbsNVDZr9E8PD2IOQ3hSccRnB6xWzw==", - "license": "MIT", - "dependencies": { - "motion-utils": "^12.12.1" - } - }, - "node_modules/motion-utils": { - "version": "12.12.1", - "resolved": "https://registry.npmjs.org/motion-utils/-/motion-utils-12.12.1.tgz", - "integrity": "sha512-f9qiqUHm7hWSLlNW8gS9pisnsN7CRFRD58vNjptKdsqFLpkVnX00TNeD6Q0d27V9KzT7ySFyK1TZ/DShfVOv6w==", - "license": "MIT" - }, "node_modules/ms": { "version": "2.1.3", "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz", @@ -7075,75 +6570,6 @@ "react": ">=18" } }, - "node_modules/react-remove-scroll": { - "version": "2.7.1", - "resolved": "https://registry.npmjs.org/react-remove-scroll/-/react-remove-scroll-2.7.1.tgz", - "integrity": "sha512-HpMh8+oahmIdOuS5aFKKY6Pyog+FNaZV/XyJOq7b4YFwsFHe5yYfdbIalI4k3vU2nSDql7YskmUseHsRrJqIPA==", - "license": "MIT", - "dependencies": { - "react-remove-scroll-bar": "^2.3.7", - "react-style-singleton": "^2.2.3", - "tslib": "^2.1.0", - "use-callback-ref": "^1.3.3", - "use-sidecar": "^1.1.3" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/react-remove-scroll-bar": { - "version": "2.3.8", - "resolved": "https://registry.npmjs.org/react-remove-scroll-bar/-/react-remove-scroll-bar-2.3.8.tgz", - "integrity": "sha512-9r+yi9+mgU33AKcj6IbT9oRCO78WriSj6t/cF8DWBZJ9aOGPOTEDvdUDz1FwKim7QXWwmHqtdHnRJfhAxEG46Q==", - "license": "MIT", - "dependencies": { - "react-style-singleton": "^2.2.2", - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/react-style-singleton": { - "version": "2.2.3", - "resolved": "https://registry.npmjs.org/react-style-singleton/-/react-style-singleton-2.2.3.tgz", - "integrity": "sha512-b6jSvxvVnyptAiLjbkWLE/lOnR4lfTtDAl+eUC7RZy+QQWc6wRzIV2CE6xBuMmDxc2qIihtDCZD5NPOFl7fRBQ==", - "license": "MIT", - "dependencies": { - "get-nonce": "^1.0.0", - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, "node_modules/reflect.getprototypeof": { "version": "1.0.10", "resolved": "https://registry.npmjs.org/reflect.getprototypeof/-/reflect.getprototypeof-1.0.10.tgz", @@ -7188,6 +6614,21 @@ "url": "https://github.com/sponsors/ljharb" } }, + "node_modules/remark-breaks": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/remark-breaks/-/remark-breaks-4.0.0.tgz", + "integrity": "sha512-IjEjJOkH4FuJvHZVIW0QCDWxcG96kCq7An/KVH2NfJe6rKZU2AsHeB3OEjPNRxi4QC34Xdx7I2KGYn6IpT7gxQ==", + "license": "MIT", + "dependencies": { + "@types/mdast": "^4.0.0", + "mdast-util-newline-to-break": "^2.0.0", + "unified": "^11.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/unified" + } + }, "node_modules/remark-gfm": { "version": "4.0.1", "resolved": "https://registry.npmjs.org/remark-gfm/-/remark-gfm-4.0.1.tgz", @@ -8295,49 +7736,6 @@ "punycode": "^2.1.0" } }, - "node_modules/use-callback-ref": { - "version": "1.3.3", - "resolved": "https://registry.npmjs.org/use-callback-ref/-/use-callback-ref-1.3.3.tgz", - "integrity": "sha512-jQL3lRnocaFtu3V00JToYz/4QkNWswxijDaCVNZRiRTO3HQDLsdu1ZtmIUvV4yPp+rvWm5j0y0TG/S61cuijTg==", - "license": "MIT", - "dependencies": { - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/use-sidecar": { - "version": "1.1.3", - "resolved": "https://registry.npmjs.org/use-sidecar/-/use-sidecar-1.1.3.tgz", - "integrity": "sha512-Fedw0aZvkhynoPYlA5WXrMCAMm+nSWdZt6lzJQ7Ok8S6Q+VsHmHpRWndVRJ8Be0ZbkfPc5LRYH+5XrzXcEeLRQ==", - "license": "MIT", - "dependencies": { - "detect-node-es": "^1.1.0", - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, "node_modules/use-sync-external-store": { "version": "1.5.0", "resolved": "https://registry.npmjs.org/use-sync-external-store/-/use-sync-external-store-1.5.0.tgz", diff --git a/package.json b/package.json index 30219fb5..56fe8e26 100644 --- a/package.json +++ b/package.json @@ -10,18 +10,16 @@ }, "dependencies": { "@radix-ui/react-avatar": "^1.1.10", - "@radix-ui/react-dropdown-menu": "^2.1.15", "@radix-ui/react-scroll-area": "^1.2.9", - "@radix-ui/react-separator": "^1.1.7", "@radix-ui/react-slot": "^1.2.3", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", - "framer-motion": "^12.16.0", "lucide-react": "^0.513.0", "next": "15.3.3", "react": "^19.0.0", "react-dom": "^19.0.0", "react-markdown": "^10.1.0", + "remark-breaks": "^4.0.0", "remark-gfm": "^4.0.1", "tailwind-merge": "^3.3.0" }, diff --git a/public/.gitkeep b/public/.gitkeep new file mode 100644 index 00000000..e69de29b diff --git a/public/file.svg b/public/file.svg deleted file mode 100644 index 004145cd..00000000 --- a/public/file.svg +++ /dev/null @@ -1 +0,0 @@ -<svg fill="none" viewBox="0 0 16 16" xmlns="http://www.w3.org/2000/svg"><path d="M14.5 13.5V5.41a1 1 0 0 0-.3-.7L9.8.29A1 1 0 0 0 9.08 0H1.5v13.5A2.5 2.5 0 0 0 4 16h8a2.5 2.5 0 0 0 2.5-2.5m-1.5 0v-7H8v-5H3v12a1 1 0 0 0 1 1h8a1 1 0 0 0 1-1M9.5 5V2.12L12.38 5zM5.13 5h-.62v1.25h2.12V5zm-.62 3h7.12v1.25H4.5zm.62 3h-.62v1.25h7.12V11z" clip-rule="evenodd" fill="#666" fill-rule="evenodd"/></svg> \ No newline at end of file diff --git a/public/globe.svg b/public/globe.svg deleted file mode 100644 index 567f17b0..00000000 --- a/public/globe.svg +++ /dev/null @@ -1 +0,0 @@ -<svg fill="none" xmlns="http://www.w3.org/2000/svg" viewBox="0 0 16 16"><g clip-path="url(#a)"><path fill-rule="evenodd" clip-rule="evenodd" d="M10.27 14.1a6.5 6.5 0 0 0 3.67-3.45q-1.24.21-2.7.34-.31 1.83-.97 3.1M8 16A8 8 0 1 0 8 0a8 8 0 0 0 0 16m.48-1.52a7 7 0 0 1-.96 0H7.5a4 4 0 0 1-.84-1.32q-.38-.89-.63-2.08a40 40 0 0 0 3.92 0q-.25 1.2-.63 2.08a4 4 0 0 1-.84 1.31zm2.94-4.76q1.66-.15 2.95-.43a7 7 0 0 0 0-2.58q-1.3-.27-2.95-.43a18 18 0 0 1 0 3.44m-1.27-3.54a17 17 0 0 1 0 3.64 39 39 0 0 1-4.3 0 17 17 0 0 1 0-3.64 39 39 0 0 1 4.3 0m1.1-1.17q1.45.13 2.69.34a6.5 6.5 0 0 0-3.67-3.44q.65 1.26.98 3.1M8.48 1.5l.01.02q.41.37.84 1.31.38.89.63 2.08a40 40 0 0 0-3.92 0q.25-1.2.63-2.08a4 4 0 0 1 .85-1.32 7 7 0 0 1 .96 0m-2.75.4a6.5 6.5 0 0 0-3.67 3.44 29 29 0 0 1 2.7-.34q.31-1.83.97-3.1M4.58 6.28q-1.66.16-2.95.43a7 7 0 0 0 0 2.58q1.3.27 2.95.43a18 18 0 0 1 0-3.44m.17 4.71q-1.45-.12-2.69-.34a6.5 6.5 0 0 0 3.67 3.44q-.65-1.27-.98-3.1" fill="#666"/></g><defs><clipPath id="a"><path fill="#fff" d="M0 0h16v16H0z"/></clipPath></defs></svg> \ No newline at end of file diff --git a/public/next.svg b/public/next.svg deleted file mode 100644 index 5174b28c..00000000 --- a/public/next.svg +++ /dev/null @@ -1 +0,0 @@ -<svg xmlns="http://www.w3.org/2000/svg" fill="none" viewBox="0 0 394 80"><path fill="#000" d="M262 0h68.5v12.7h-27.2v66.6h-13.6V12.7H262V0ZM149 0v12.7H94v20.4h44.3v12.6H94v21h55v12.6H80.5V0h68.7zm34.3 0h-17.8l63.8 79.4h17.9l-32-39.7 32-39.6h-17.9l-23 28.6-23-28.6zm18.3 56.7-9-11-27.1 33.7h17.8l18.3-22.7z"/><path fill="#000" d="M81 79.3 17 0H0v79.3h13.6V17l50.2 62.3H81Zm252.6-.4c-1 0-1.8-.4-2.5-1s-1.1-1.6-1.1-2.6.3-1.8 1-2.5 1.6-1 2.6-1 1.8.3 2.5 1a3.4 3.4 0 0 1 .6 4.3 3.7 3.7 0 0 1-3 1.8zm23.2-33.5h6v23.3c0 2.1-.4 4-1.3 5.5a9.1 9.1 0 0 1-3.8 3.5c-1.6.8-3.5 1.3-5.7 1.3-2 0-3.7-.4-5.3-1s-2.8-1.8-3.7-3.2c-.9-1.3-1.4-3-1.4-5h6c.1.8.3 1.6.7 2.2s1 1.2 1.6 1.5c.7.4 1.5.5 2.4.5 1 0 1.8-.2 2.4-.6a4 4 0 0 0 1.6-1.8c.3-.8.5-1.8.5-3V45.5zm30.9 9.1a4.4 4.4 0 0 0-2-3.3 7.5 7.5 0 0 0-4.3-1.1c-1.3 0-2.4.2-3.3.5-.9.4-1.6 1-2 1.6a3.5 3.5 0 0 0-.3 4c.3.5.7.9 1.3 1.2l1.8 1 2 .5 3.2.8c1.3.3 2.5.7 3.7 1.2a13 13 0 0 1 3.2 1.8 8.1 8.1 0 0 1 3 6.5c0 2-.5 3.7-1.5 5.1a10 10 0 0 1-4.4 3.5c-1.8.8-4.1 1.2-6.8 1.2-2.6 0-4.9-.4-6.8-1.2-2-.8-3.4-2-4.5-3.5a10 10 0 0 1-1.7-5.6h6a5 5 0 0 0 3.5 4.6c1 .4 2.2.6 3.4.6 1.3 0 2.5-.2 3.5-.6 1-.4 1.8-1 2.4-1.7a4 4 0 0 0 .8-2.4c0-.9-.2-1.6-.7-2.2a11 11 0 0 0-2.1-1.4l-3.2-1-3.8-1c-2.8-.7-5-1.7-6.6-3.2a7.2 7.2 0 0 1-2.4-5.7 8 8 0 0 1 1.7-5 10 10 0 0 1 4.3-3.5c2-.8 4-1.2 6.4-1.2 2.3 0 4.4.4 6.2 1.2 1.8.8 3.2 2 4.3 3.4 1 1.4 1.5 3 1.5 5h-5.8z"/></svg> \ No newline at end of file diff --git a/public/vercel.svg b/public/vercel.svg deleted file mode 100644 index 77053960..00000000 --- a/public/vercel.svg +++ /dev/null @@ -1 +0,0 @@ -<svg fill="none" xmlns="http://www.w3.org/2000/svg" viewBox="0 0 1155 1000"><path d="m577.3 0 577.4 1000H0z" fill="#fff"/></svg> \ No newline at end of file diff --git a/public/window.svg b/public/window.svg deleted file mode 100644 index b2b2a44f..00000000 --- a/public/window.svg +++ /dev/null @@ -1 +0,0 @@ -<svg fill="none" xmlns="http://www.w3.org/2000/svg" viewBox="0 0 16 16"><path fill-rule="evenodd" clip-rule="evenodd" d="M1.5 2.5h13v10a1 1 0 0 1-1 1h-11a1 1 0 0 1-1-1zM0 1h16v11.5a2.5 2.5 0 0 1-2.5 2.5h-11A2.5 2.5 0 0 1 0 12.5zm3.75 4.5a.75.75 0 1 0 0-1.5.75.75 0 0 0 0 1.5M7 4.75a.75.75 0 1 1-1.5 0 .75.75 0 0 1 1.5 0m1.75.75a.75.75 0 1 0 0-1.5.75.75 0 0 0 0 1.5" fill="#666"/></svg> \ No newline at end of file diff --git a/rag_system/DOCUMENTATION.md b/rag_system/DOCUMENTATION.md index bf240d59..b2117786 100644 --- a/rag_system/DOCUMENTATION.md +++ b/rag_system/DOCUMENTATION.md @@ -1,62 +1,464 @@ # RAG System Documentation -This document provides a detailed overview of the RAG (Retrieval-Augmented Generation) system, its architecture, and how to use it. +Reference for the `rag_system` package: pipelines, configuration keys, the HTTP API and +the CLI. For the shorter tour of the package, see [`README.md`](README.md). -## System Overview +Everything below is described as the code behaves today. Where a knob exists but nothing +reads it, that is called out explicitly rather than glossed over. -This RAG system is a sophisticated, multimodal question-answering system designed to work with a variety of documents. It can understand and process both the text and the visual layout of documents, and it uses a knowledge graph to understand the relationships between the entities in the documents. +## 1. Where this runs -The system is built around an agentic workflow that allows it to: +`rag_system` is one process in a four-process system: -* **Decompose complex questions** into smaller, more manageable sub-questions. -* **Triage queries** to determine if they can be answered directly or if they require retrieval from the knowledge base. -* **Verify answers** against the retrieved context to ensure they are accurate and supported by the documents. +``` +Next.js frontend :3000 โ”€โ”€โ–บ backend gateway :8000 โ”€โ”€โ–บ RAG API :8001 โ”€โ”€โ–บ Ollama :11434 + โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ /chat/stream (SSE) โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ–บ +``` -## Architecture +The RAG API (`rag_system/api_server.py`) is a standard-library `http.server` running on a +plain `socketserver.TCPServer`, so **requests are handled one at a time**. A long chat or +indexing call blocks every other request to port 8001 until it finishes. Plan for a single +concurrent user per RAG API process. -The system is composed of two main pipelines: an indexing pipeline and a retrieval pipeline. +The API opens the same SQLite file as the backend gateway (`backend/chat_data.db`, or +`DB_PATH`), but only to read sessionโ†’index links and to read/write index metadata. +**Chat message rows are written exclusively by `backend/server.py`.** The frontend's +streaming path posts straight to `:8001/chat/stream`; the stream itself persists nothing, +and the browser saves the completed turn through the gateway +(`POST :8000/sessions/<id>/messages/save`) once the `complete` event arrives. -### Indexing Pipeline +## 2. Indexing pipeline -The indexing pipeline is responsible for processing the documents and building the knowledge base. It performs the following steps: +`rag_system/pipelines/indexing_pipeline.py`. Public entry point: -1. **Text Extraction**: The pipeline uses `PyMuPDF` to extract the text from each page of the PDF documents, preserving the original layout. -2. **Text Embedding**: The extracted text is then passed to a text embedding model (`Qwen/Qwen3-Embedding-0.6B`) to create numerical vector representations of the text. -3. **Knowledge Graph Creation**: The text is also passed to a `GraphExtractor` that uses a large language model (`qwen2.5vl:7b`) to extract entities and their relationships. This information is then used to build a knowledge graph, which is stored as a `.gml` file. -4. **Indexing**: The text embeddings and the knowledge graph are then stored in a LanceDB database. +```python +IndexingPipeline(config, llm_client, ollama_config).run(file_paths: list[str]) +``` -### Retrieval Pipeline +`run()` also accepts the legacy keyword alias `documents=`. There is no +`process_documents()` method. -The retrieval pipeline is responsible for answering user queries. It uses an agentic workflow that includes the following steps: +### Steps -1. **Triage**: The agent first triages the user's query to determine if it can be answered directly or if it requires retrieval from the knowledge base. -2. **Query Decomposition**: If the query is complex, the agent uses a `QueryDecomposer` to break it down into smaller, more manageable sub-questions. -3. **Retrieval**: The agent then uses a `MultiVectorRetriever` and a `GraphRetriever` to retrieve relevant information from the knowledge base. -4. **Verification**: The retrieved context is then passed to a `Verifier` that uses an LLM to check if the context is sufficient to answer the query. -5. **Synthesis**: Finally, the agent uses an LLM to synthesize a final answer from the verified context. +1. **Conversion** โ€” `ingestion/document_converter.py`. Docling converts the file to a + single Markdown string plus the `DoclingDocument` object. + - `.pdf`, `.docx`, `.html`, `.htm`, `.md` go through Docling; `.txt` is read + directly and wrapped in a fenced block. + - PDFs are first probed with PyMuPDF (`fitz`). If any page yields text, the no-OCR + converter is used; otherwise the OCR converter is used. + - OCR engine selection is dynamic: `OcrMacOptions` (macOS only), then `EasyOcrOptions`, + `RapidOcrOptions`, `TesseractOcrOptions`, `TesseractCliOcrOptions` โ€” the first one + whose backend module or binary is actually installed wins. If none is available, + Docling's default OCR settings are used and a message is printed. + - The three converters (no-OCR, OCR, general) are built independently, so a failing + OCR engine does not disable the other paths. -## API Endpoints +2. **Chunking** โ€” `chunker_mode` selects the chunker. + - `docling` (default): `ingestion/docling_chunker.py`. Walks the DoclingDocument + element tree in reading order, emits tables as atomic Markdown chunks, tracks the + heading path, and token-packs paragraphs up to `chunk_size` using the embedding + model's tokenizer. Each chunk records `heading_path`, `heading_level`, + `block_type` and (when available) `page`. If the tree walk fails it falls back to + splitting the exported Markdown with `overlap_sentences` sentences of overlap + (default 1). + - `legacy`: `ingestion/chunking.py::MarkdownRecursiveChunker`. Recursively splits on + `\n## `, `\n### `, `\n#### `, code fences and blank lines, measuring size with the + embedding model's tokenizer, with `max_chunk_size = chunk_size` and + `min_chunk_size = max(1, chunk_size // 4)`. + - If the Docling chunker fails to initialise, the pipeline falls back to the legacy + chunker automatically. + - Each chunk gets a sequential `metadata.chunk_index` within its document. -The system provides the following command-line endpoints: +3. **Document overview** โ€” `indexing/overview_builder.py`, on unless + `overview.enabled` is `false`. The enrichment model summarises the first + `overview_first_n_chunks` (or `overview.max_chunks`, default 5) chunks into one + paragraph, appended as JSONL to `overview_path` + (default `index_store/overviews/overviews.jsonl`). These overviews are what the query + router reads later. Failures here are logged and do not abort indexing. -* `index`: This endpoint runs the indexing pipeline to process the documents and build the knowledge base. -* `chat`: This endpoint runs the retrieval pipeline to answer a user's query. -* `show_graph`: This endpoint displays the knowledge graph in a human-readable format and also provides a visual representation of the graph. +4. **Contextual enrichment** โ€” `indexing/contextualizer.py`, when + `contextual_enricher.enabled`. For each chunk the enrichment model writes a 2โ€“5 + sentence summary of the surrounding `window_size` chunks; the summary is prepended to + the chunk text (`"Context: โ€ฆ\n\n---\n\n<original>"`) and the untouched text is stored + in `metadata.original_text`. The enriched text is what gets embedded. -### Usage +5. **Embedding and indexing** โ€” `indexing/representations.py` + + `indexing/embedders.py`. Chunks are embedded in batches of + `indexing.embedding_batch_size` and written to LanceDB. The Arrow schema is + `vector` (fixed-size float32 list), `text`, `chunk_id`, `document_id`, `chunk_index`, + `metadata` (JSON string of the whole chunk). + - The vector width comes from the embeddings the model produced. Appending to a table + whose stored width differs raises + `Table '<name>' stores N-dim vectors but the current embedding model produced M-dim + vectors` โ€” re-index instead. + - Chunks whose vector contains NaN/Inf are skipped with a warning. + - After the append, a **native LanceDB FTS index** is created on `text` (index name + `text_idx`, `use_tantivy=False`) unless `text_idx` or the historical `fts_text` + already exists. There is no separate BM25 library or sidecar index. -To run the system, use the following commands: +6. **Late chunking** (optional) โ€” `indexing/latechunk.py`. The whole document is fed to + the embedding model once, per-token hidden states are mean-pooled inside each chunk's + character span, and the resulting vectors are written to a sibling table: + `latechunk.lancedb_table_name` if set, otherwise + `f"{table_name}{latechunk.table_suffix}"` with `table_suffix` defaulting to `_lc`. + Documents whose vector count does not match their chunk count are skipped. -```bash -# Activate the virtual environment -source rag_system/rag_venv/bin/activate +There is no step 7. Knowledge-graph extraction (`indexing/graph_extractor.py`, the +`retrieval.graph.*` keys, the NetworkX `.gml` writer) was **removed on 2026-08-09** +(roadmap item 2.5): unreachable in every shipped profile, and evidence-negative โ€” +GraphRAG loses on single-hop retrieval, its multi-hop gains are contested, and it costs +41โ€“57x at indexing and up to ~377x in query tokens +(`Documentation/research/academic-evidence-2026.md` ยง6). + +## 3. Retrieval: agent and pipeline + +### 3.1 Agent (`rag_system/agent/loop.py`) + +```python +Agent.run( + query, + table_name=None, session_id=None, + compose_sub_answers=None, query_decompose=None, ai_rerank=None, + context_expand=None, verify=None, + retrieval_k=None, context_window_size=None, reranker_top_k=None, + retrieval_mode=None, force_rag=False, event_callback=None, +) -> {"answer": str, "source_documents": list[dict]} +``` + +`run()` is a synchronous wrapper around `_run_async()`. Every `None` argument leaves the +profile value in place; anything else is written into the live retrieval-pipeline config. + +**Routing order.** + +1. If `force_rag` is true, triage is skipped and the query type is pinned to `rag_query`. +2. Otherwise the overview router runs first: the enrichment model is shown up to 40 + document overviews and returns `{"category": "direct_answer" | "rag_query"}`. If no + overviews are loaded it returns nothing and routing continues. +3. If the session already has history, the query is treated as a follow-up โ†’ + `rag_query`. +4. Otherwise an LLM triage prompt picks `rag_query` or `direct_answer`. + A JSON parse failure defaults to `rag_query`. + +`Agent._normalize_triage()` is applied to every verdict: anything that is not an explicit +`direct_answer` becomes `rag_query`. That is what catches a small utility model still +emitting the retired `graph_query` label (the graph module was removed on 2026-08-09). + +**Semantic cache.** For non-`direct_answer` queries the raw query is embedded and compared +against a `TTLCache(maxsize=100, ttl=300)` of previous results. A cosine similarity โ‰ฅ +`semantic_cache_threshold` (0.98) is a hit. With `cache_scope` at its default `session`, +entries from other sessions are skipped; setting it to `global` shares answers โ€” including +document-derived ones โ€” across sessions. + +**Query decomposition.** When enabled, `retrieval/query_transformer.py::QueryDecomposer` +splits the raw query (plus the last 5 turns for pronoun resolution) into at most +`query_decomposition.max_sub_queries` (default 10) sub-queries. What happens next depends +on `compose_from_sub_answers` (default `false`) and `pooled_first_stage` (default `true` +since arm H, 2026-08-15): + +* **`compose_from_sub_answers: false` + `pooled_first_stage: true`** (shipped default) โ€” + the first stage runs **per sub-query**, the candidates are pooled and de-duplicated, + then get ONE rerank pass and ONE synthesis over the union context + (`_pooled_first_stage` in `pipelines/retrieval_pipeline.py`). This replaced N rerank + passes, N synthesis calls and the compose step, where multi-hop facts were measurably + lost (arm E). +* **`compose_from_sub_answers: false` + `pooled_first_stage: false`** โ€” the first stage + runs **once, on the full original query**, and the sub-queries are applied at the + **rerank** stage instead: every candidate is scored against every sub-query and the + scores combined with `query_decomposition.rerank_aggregate` (`"mean"` default, or + `"max"`). With reranking off there is no rerank stage, so the sub-queries go unused. +* **`compose_from_sub_answers: true`** โ€” one full `RetrievalPipeline.run()` per sub-query, + in parallel on up to 3 worker threads, then the generation model composes one answer from + the sub-answers. Available as an option; no longer the shipped default. +* One sub-query after decomposition takes the direct path with the resolved query. + +**Verification.** When `verification.enabled` (or the per-request `verify` flag) is true +*and* the result has source documents, `agent/verifier.py` grades groundedness with the +enrichment model and appends ` [Confidence: N%]` to the answer string, plus +` [Warning: Low confidence. Groundedness: <bool>]` when the answer is not grounded or the +score is under 50. A score of 0 (parse failure) appends nothing. There is no separate +`confidence` field in the response. + +Setting `verification.model` (or the `VERIFIER_MODEL` env var) to a HuggingFace model name +swaps the LLM prompt for a local NLI/verifier model scored per answer sentence against the +evidence (roadmap 2.4, off by default). `Documentation/verifier.md` has the availability +findings and the reasons `[Confidence: N%]` is UX rather than a calibrated measurement. + +**History.** The agent keeps an in-process `LRUCache(maxsize=100)` of per-session +`{query, answer}` turns. This is not persisted and is lost on restart. + +### 3.2 Retrieval pipeline (`rag_system/pipelines/retrieval_pipeline.py`) + +`run(query, table_name=None, window_size_override=None, event_callback=None)`: + +1. **Retrieve** โ€” `retrieval/retrievers.py::MultiVectorRetriever.retrieve(text_query, + table_name, k, search_type)`: + - `hybrid` (default): the FTS leg and the vector leg each fetch `k` rows in parallel + and are fused with **reciprocal rank fusion** (`1/(60 + rank)` per leg). There are + no tunable leg weights. + - `vector_only` / `fts_only`: a single leg; the ordering is LanceDB's own. + - Single-word FTS queries are expanded to `word* OR word~` for prefix/fuzzy recall. + - Every returned document carries a finite, higher-is-better `score`: the BM25 score + for `fts_only`, `1/(1+distance)` for `vector_only`, the RRF score for `hybrid`. + `bm25` and `_distance` are only present when that leg actually matched. + - An unknown mode logs a warning and degrades to `hybrid`. +2. **Late-chunk leg** โ€” if late chunking is enabled, the same query also runs against the + `_lc` table and those hits are appended to the candidate list. Then "late-chunk + merging" runs over every candidate: each chunk's text is replaced by itself plus its + ยฑ1 neighbours from the main table, joined in `chunk_index` order. +2b. **Evidence-sufficiency retry** โ€” if `retrieval.retry.enabled` (true in `default`, false + in `fast`) and the first pass found weak evidence, the query is reformulated once on the + enrichment model and steps 1โ€“3 run again; the better of the two result sets is kept. The + signal is the *contrast* between the top candidate's cosine similarity and the + background of the rest, not the raw top similarity (which measured anti-correlated with + success), and it is only available on L2-normalized v4+ tables. See + `Documentation/retrieval_pipeline.md` ยง2b. +3. **Rerank** โ€” if `reranker.enabled` and a reranker loaded. `strategy: "rerankers-lib"` + (the default) loads the model through the `rerankers` library with + `model_type` (default `cross-encoder`); any other strategy uses the local + `rerankers/reranker.py::CrossEncoderReranker`. Results are trimmed to + `reranker.top_k` (or `reranker.top_percent` of the candidate count). **If the model + cannot be loaded the pipeline prints a warning and skips reranking** rather than + failing. +4. **Context expansion** โ€” when the effective window is > 0, each surviving chunk pulls + `chunk_index ยฑ window_size` siblings from the same `document_id` via a LanceDB metadata + filter. Results are re-sorted by `rerank_score`, then `_distance`, then `score`. If any + chunk carries a `rerank_score`, non-reranked chunks are dropped. +5. **Sentence pruning** โ€” when `provence.enabled`, `rerankers/sentence_pruner.py` runs + `naver/provence-reranker-debertav3-v1` over each chunk at `provence.threshold` + (default 0.1) and chunks pruned to empty are dropped. If the Provence weights cannot be + loaded, pruning is a no-op. +6. **Synthesis** โ€” the generation model streams the final answer from the surviving + chunks. `_distance` and `vector` are stripped from the returned documents and + NaN/Inf numerics are nulled so the payload is JSON-safe. + +Returned source documents have the shape +`{chunk_id, text, score, document_id, chunk_index, metadata}` plus `bm25` and/or +`rerank_score` when those stages ran. `metadata` is the chunk record that was serialised +into the LanceDB `metadata` column at index time, with `document_id` and `chunk_index` +filled in from the row. + +## 4. Configuration reference + +Defined in `rag_system/main.py`. `factory.get_pipeline_config(mode)` hands out a deep copy, +so per-request overrides never mutate the master dictionaries. + +### 4.1 Model configuration + +| Object | Keys | Env overrides | +| --- | --- | --- | +| `LLM_BACKEND` | `ollama` (default) or `watsonx` | `LLM_BACKEND` | +| `OLLAMA_CONFIG` | `host`, `generation_model`, `enrichment_model` | `OLLAMA_HOST`, `GENERATION_MODEL`, `ENRICHMENT_MODEL` | +| `WATSONX_CONFIG` | `api_key`, `project_id`, `url`, `generation_model`, `enrichment_model` | `WATSONX_API_KEY`, `WATSONX_PROJECT_ID`, `WATSONX_URL`, `WATSONX_GENERATION_MODEL`, `WATSONX_ENRICHMENT_MODEL` | +| `EXTERNAL_MODELS` | `embedding_model`, `reranker_model` | `EMBEDDING_MODEL`, `RERANKER_MODEL` | + +### 4.2 Pipeline profile keys + +`PIPELINE_CONFIGS` contains exactly two profiles: `default` and `fast`. + +| Key | Read by | Notes | +| --- | --- | --- | +| `storage.lancedb_uri` | both pipelines | LanceDB directory (`./lancedb`). Also accepted as `storage.db_path` / `storage.lancedb_path`. | +| `storage.text_table_name` | both pipelines | Default table (`text_pages_v4`). | +| `retrieval.search_type` | `RetrievalPipeline._retrieval_mode` | `hybrid`, `vector_only`, `fts_only`. | +| `retrieval.dense.enabled` | both pipelines | `false` skips vector indexing and disables retrieval entirely โ€” `MultiVectorRetriever` owns both the FTS and the vector leg, so no retriever is built at all. | +| `retrieval.dense.lancedb_table_name` | indexing | Fallback table name when `storage.text_table_name` is unset. | +| `retrieval.latechunk.enabled` | both pipelines | Also accepted under `retrievers.latechunk` / `retrieval.late_chunking`. | +| `retrieval.latechunk.table_suffix` / `.lancedb_table_name` | both pipelines | Late-chunk table name; suffix defaults to `_lc` on both sides. | +| `embedding_model_name` | both pipelines | Required โ€” the retrieval pipeline raises if it is missing rather than guessing a dimension. | +| `reranker.enabled` | retrieval | | +| `reranker.strategy` | retrieval | `rerankers-lib` (default) or anything else โ†’ local `CrossEncoderReranker`. | +| `reranker.model_type` | retrieval | `rerankers` library model type, default `cross-encoder`. Use `colbert` for late-interaction models. | +| `reranker.model_name` | retrieval | Missing value logs a warning and skips reranking. | +| `reranker.top_k` / `reranker.top_percent` | retrieval | `top_percent` (0โ€“1) wins when set. | +| `query_decomposition.enabled` | agent | | +| `query_decomposition.compose_from_sub_answers` | agent | Default `false`; with `pooled_first_stage: true` (also the default) the first stage runs per sub-query and candidates are pooled for one rerank + one synthesis. | +| `query_decomposition.max_sub_queries` | agent | Default 10; not present in the shipped profiles. | +| `query_decomposition.rerank_aggregate` | retrieval | `mean` (default) or `max`; how per-sub-query rerank scores combine. Only read when reranking runs with sub-queries. | +| `retrieval.retry.enabled` / `.min_top_score` / `.max_attempts` | retrieval | Evidence-sufficiency retry. `true` / `0.12` / `1` in `default`; disabled in `fast`. Also accepted under `retrievers.retry`. | +| `retrieval.retry.min_rerank_score` | retrieval | Threshold used instead of `min_top_score` when the reranker produced a 0โ€“1 probability. Defaults to `min_top_score`. | +| `verification.enabled` | agent | | +| `verification.model` | agent | HuggingFace model name for the local verifier; unset โ‡’ the LLM-prompt verifier. Also settable as `VERIFIER_MODEL`. | +| `verification.threshold` | agent | Default `0.5`. Local verifier only. | +| `retrieval_k` | retrieval | Rows fetched per leg; the fused list is truncated to the same value. | +| `context_window_size` | retrieval | `0` disables context expansion (the value in both shipped profiles). | +| `semantic_cache_threshold` | agent | Cosine similarity for a cache hit (0.98). | +| `cache_scope` | agent | `session` (default) or `global`. | +| `contextual_enricher.enabled` / `.window_size` | indexing | | +| `enrich_model` / `enrichment_model_name` | indexing | Per-index override of `OLLAMA_CONFIG.enrichment_model`. | +| `overview.enabled` / `.model` / `.max_chunks` | indexing | Not present in the shipped profiles; defaults are enabled, enrichment model, 5 chunks. | +| `overview_model_name` / `overview_first_n_chunks` / `overview_path` | indexing | Top-level equivalents, set per request by the API. | +| `chunker_mode` | indexing | `docling` (default) or `legacy`. | +| `chunking.chunk_size` | indexing | Token budget. Also accepted as top-level `chunk_size` or `max_tokens`. **Default when unset: 1500**; the HTTP `/index` endpoint sends 512. | +| `overlap_sentences` | indexing | Docling chunker sentence overlap, default 1. | +| `indexing.embedding_batch_size` / `.enrichment_batch_size` | indexing | | +| `provence.enabled` / `.threshold` | retrieval | Not in the profiles; set per request by the API. Threshold default 0.1. | + +Keys that exist in the shipped profiles but that **nothing reads**: `description`, +`reranker.type`, and `indexing.enable_progress_tracking`. + +`retrieval.graph.*` and `graph_strategy.*` no longer exist โ€” they are ignored if present +in a hand-written config (see the removal note in ยง2). + +There is no `dense.weight`/`denseWeight`, no `bm25_path` or other BM25 index path, no +`fallback_reranker`, no `vision_model_name`, and no `chunk_overlap` โ€” the sparse leg is +LanceDB's native FTS, hybrid fusion is weight-free, and chunk overlap was never applied by +either chunker. -# Index the documents -python rag_system/main.py index +## 5. HTTP API -# Ask a question -python rag_system/main.py chat "Your question here" +`rag_system/api_server.py`, default port 8001. Start it with +`python -m rag_system.api_server` or `python -m rag_system.main api --port 8001`. The +profile it loads comes from `RAG_CONFIG_MODE` (default `default`). -# Show the knowledge graph -python rag_system/main.py show_graph +All responses are JSON with `Access-Control-Allow-Origin: *`. Errors are +`{"error": "<message>"}` with a 4xx/5xx status. `OPTIONS` on any path returns the CORS +preflight headers (`GET, POST, OPTIONS`); any unmatched path returns +`{"error": "Not Found"}` with 404. + +**Key casing.** Every request body is normalised once at parse time: `camelCase` keys are +converted to `snake_case`, so `rerankerTopK` and `reranker_top_k` land in the same place. +An explicit `snake_case` value always wins over its camelCase twin. `overview_model` / +`overviewModel` are additionally aliased to `overview_model_name`. Unknown fields are +ignored. + +### `GET /health` + +```json +{"status": "ok"} +``` + +### `GET /models` + +```json +{"generation_models": ["..."], "embedding_models": ["..."]} +``` + +With `LLM_BACKEND=ollama` the list comes from `GET {OLLAMA_HOST}/api/tags` (5 s timeout); +tags whose name contains `embed`, `bge` or `embedding` are classified as embedding models +and the rest as generation models. With `LLM_BACKEND=watsonx` the generation list is the +two configured granite ids. `embedding_models` always also contains the currently +configured embedding model plus `microsoft/harrier-oss-v1-0.6b` and +`Qwen/Qwen3-Embedding-4B`, `-0.6B` and `-8B`. + +### `POST /chat` + +| Field | Type | Default | Behaviour | +| --- | --- | --- | --- | +| `query` | string | โ€” | **Required**; 400 if missing. | +| `session_id` | string | โ€“ | Loads that session's linked index (table name, embedding model, overviews) and its in-memory history. | +| `table_name` | string | โ€“ | Explicit LanceDB table; otherwise resolved from `session_id`, otherwise the profile default. | +| `model` | string | โ€“ | Per-request generation model, applied for this request only and restored afterwards. Ignored with a warning when the id is not valid for the active backend (watsonx ids contain `/`, Ollama tags do not). | +| `retrieval_mode` (alias `search_type`) | string | profile | `hybrid`, `vector_only`, `fts_only`. **Any other value is a 400.** | +| `force_rag` | bool | `false` | Skips triage and forces the RAG path; all other toggles still apply. | +| `query_decompose` | bool | profile | | +| `compose_sub_answers` | bool | profile | | +| `ai_rerank` | bool | profile | Toggles `reranker.enabled`. | +| `context_expand` | bool | profile | `false` forces the expansion window to 0. | +| `verify` | bool | profile | | +| `retrieval_k` | int | `20` | | +| `context_window_size` | int | `1` | Note: the profiles ship `0`, so an HTTP request expands context by default while the CLI does not. | +| `reranker_top_k` | int | `10` | | +| `provence_prune` | bool | off | Enables Provence sentence pruning. | +| `provence_threshold` | float | `0.1` | | + +Response: + +```json +{ + "answer": "โ€ฆ", + "source_documents": [ + {"chunk_id": "โ€ฆ", "text": "โ€ฆ", "score": 0.031, "document_id": "โ€ฆ", + "chunk_index": 4, "metadata": {"โ€ฆ": "โ€ฆ"}, "rerank_score": 6.1} + ] +} +``` + +### `POST /chat/stream` + +Same request body as `/chat`. Responds with `Content-Type: text/event-stream` and one +`data: {"type": "<event>", "data": {...}}\n\n` frame per event. + +Event types: `analyze`, `direct_answer`, `decomposition`, `retrieval_started`, +`retrieval_done`, `rerank_started`, `rerank_done`, `context_expand_started`, +`context_expand_done`, `prune_started`, `prune_done`, `token`, `sub_query_token`, +`sub_query_result`, `single_query_result`, `final_answer`, `complete`, `error`. +`complete` carries the same object `/chat` would have returned and is the last frame. + +### `POST /index` + +| Field | Type | Default | Behaviour | +| --- | --- | --- | --- | +| `file_paths` | string[] | โ€” | **Required** list of absolute paths; 400 otherwise. | +| `session_id` | string | โ€“ | Resolves the target table and sets `overview_path` to `index_store/overviews/<session_id>.jsonl`. | +| `table_name` | string | โ€“ | Explicit LanceDB table. | +| `embedding_model` | string | profile | Overrides `embedding_model_name` for this build. Also written into the index metadata when `session_id` is supplied, so later queries reuse the same embedder. | +| `enrich_model` | string | profile | Model used for contextual enrichment. | +| `overview_model_name` | string | profile | Model used for document overviews. | +| `enable_latechunk` | bool | `false` | **Note:** the HTTP default is off even though the `default` profile enables late chunking. | +| `enable_enrich` | bool | `true` | | +| `window_size` | int | `2` | Contextual-enrichment window. | +| `chunk_size` | int | `512` | Token budget per chunk. | +| `enable_docling_chunk` | bool | `true` | `true` pins `chunker_mode` to `docling`. Passing `false` selects the `legacy` chunker. | +| `retrieval_mode` (alias `search_type`) | string | profile | Validated against the same three values (400 otherwise) and recorded on the index config. The mode only changes behaviour at query time. | +| `batch_size_embed` | int | `50` | | +| `batch_size_enrich` | int | `25` | | + +Response: + +```json +{ + "message": "Indexing process for 3 file(s) completed successfully.", + "table_name": "text_pages_<id>", + "latechunk": false, + "docling_chunk": true, + "indexing_config": { + "chunk_size": 512, "retrieval_mode": "hybrid", "window_size": 2, + "enable_enrich": true, "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": null, "overview_model_name": null, + "batch_size_embed": 50, "batch_size_enrich": 25 + } +} +``` + +Indexing is synchronous: the response is sent after the pipeline finishes. There is no +per-file result list and no progress endpoint. + +## 6. Command line + +Always run as a module from the repository root: + +```bash +python -m rag_system.main index <file-or-directory> [--mode default|fast] +python -m rag_system.main chat "<query>" [--mode default|fast] +python -m rag_system.main api [--port 8001] ``` + +`index` accepts a single file or walks a directory for `.pdf`, `.docx`, `.html`, `.htm`, +`.md` and `.txt`. `chat` prints the JSON result of one `Agent.run()` call. `api` is +equivalent to `python -m rag_system.api_server`. + +`python rag_system/main.py โ€ฆ` does not work (the package would not be importable). + +## 7. Storage layout + +| Path | Contents | +| --- | --- | +| `./lancedb/` | LanceDB tables. `text_pages_v4` is the profile default; the UI creates `text_pages_<index_id>` per index, and late chunking adds `<table>_lc`. Override with `LANCEDB_PATH` or `storage.lancedb_uri`. | +| `./index_store/overviews/*.jsonl` | Per-document overviews, one JSON object (`doc_id`, `overview`) per line. `overviews.jsonl` is the global fallback; per-session/index files are named `<id>.jsonl`. | +| `backend/chat_data.db` | SQLite: sessions, messages, indexes, documents and index metadata. Override with `DB_PATH`. | +| `./shared_uploads/` | Files uploaded through the web UI. | + +## 8. Operational notes + +- **Logging** โ€” `rag_system/__init__.py` configures the root logger; set `RAG_LOG_LEVEL` + (`DEBUG`/`INFO`/`WARNING`/`ERROR`, default `INFO`). Much of the pipeline still prints + progress directly to stdout. +- **Hugging Face auth** โ€” `HF_TOKEN` (or `HUGGINGFACE_HUB_TOKEN`) is picked up on import + and used to log in to the Hub. +- **Model loading** โ€” the embedder is cached per model name in-process, and the reranker + and Provence loads are guarded by locks so parallel sub-queries do not load the same + weights twice. +- **Changing the embedding model requires re-indexing.** Vector width is part of the + LanceDB table schema and a mismatch is a hard error. +- **Thread safety** โ€” `rerankers` backends are not thread-safe, so `.rank()` calls are + serialised behind a lock. The API server itself is single-threaded. diff --git a/rag_system/README.md b/rag_system/README.md index ac2e59b4..7959ac38 100644 --- a/rag_system/README.md +++ b/rag_system/README.md @@ -1,104 +1,286 @@ -# Multimodal RAG System +# RAG System -This document provides a detailed overview of the multimodal Retrieval-Augmented Generation (RAG) system implemented in this directory. The system is designed to process and understand information from PDF documents, combining both textual and visual data to answer complex queries. +This directory contains the retrieval-augmented generation engine behind localGPT: the +master configuration, the indexing and retrieval pipelines, the agent loop, and the HTTP +API that the backend gateway and the frontend talk to. + +For a deeper, key-by-key reference (config tables, HTTP request/response shapes, SSE +events) see [`DOCUMENTATION.md`](DOCUMENTATION.md) next to this file. ## 1. Overview -This RAG system is a sophisticated pipeline that leverages state-of-the-art open-source models to provide accurate, context-aware answers from a document corpus. Unlike traditional RAG systems that only process text, this implementation is fully multimodal. It extracts and indexes both text and images from PDFs, allowing a Vision Language Model (VLM) to reason over both modalities when generating a final answer. +The system is **text-only**. Documents are converted to Markdown with +[Docling](https://github.com/docling-project/docling), chunked, optionally enriched with +LLM-generated context, embedded with a local Hugging Face model, and written to +**LanceDB**. Queries are answered by combining LanceDB's native full-text search with +vector search, reranking the merged candidates, and synthesising an answer with a local +Ollama model. + +Core capabilities: + +- **Docling ingestion** โ€” PDF, DOCX, HTML/HTM and MD go through Docling; TXT is read + directly and wrapped in a fenced block. Scanned PDFs (no text layer) take an OCR + pipeline when an OCR backend is installed. +- **Hybrid retrieval** โ€” LanceDB full-text (BM25-scored, native to LanceDB) and vector + search run in parallel and are fused with reciprocal rank fusion. `vector_only` and + `fts_only` run a single leg. +- **Cross-encoder reranking** โ€” the merged candidates are reordered by a reranker loaded + through the [`rerankers`](https://github.com/AnswerDotAI/rerankers) library. +- **Agentic query handling** โ€” triage (documents vs. general knowledge), optional query + decomposition with parallel sub-query retrieval, optional answer verification. +- **Late chunking** โ€” an optional second embedding pass that encodes the whole document + and mean-pools per-chunk spans, stored in a sibling `<table>_lc` table. -The core capabilities include: -- **Multimodal Indexing**: Extracts text and images from PDFs and creates separate vector embeddings for each. -- **Hybrid Retrieval**: Combines dense vector search (for semantic similarity) with traditional keyword-based search (BM25) for robust retrieval. -- **Advanced Reranking**: Utilizes a powerful reranker model to improve the relevance of retrieved documents before they are passed to the generator. -- **VLM-Powered Synthesis**: Employs a Vision Language Model to synthesize the final answer, allowing it to analyze both the text and the images from the retrieved document chunks. +There is **no multimodal path**. Nothing produces image embeddings and no +vision-language model is invoked; PDF understanding is Docling's layout parsing plus +OCR. If you want page-image understanding you have to build it; GLM-OCR or Qwen3-VL +would be reasonable starting points, but neither is integrated today. ## 2. Architecture -The system is composed of several key Python modules that work together to form the RAG pipeline. - -### Key Modules: - -- `main.py`: The main entry point for the application. It contains the configuration for all models and pipelines and orchestrates the indexing and retrieval processes. -- `rag_system/pipelines/`: Contains the high-level orchestration for indexing and retrieval. - - `indexing_pipeline.py`: Manages the process of converting raw PDFs into indexed, searchable data. - - `retrieval_pipeline.py`: Handles the end-to-end process of taking a user query, retrieving relevant information, and generating a final answer. -- `rag_system/indexing/`: Contains all modules related to data processing and indexing. - - `multimodal.py`: Responsible for extracting text and images from PDFs and generating embeddings using the configured vision model (`colqwen2-v1.0`). - - `representations.py`: Defines the text embedding model (`Qwen2-7B-instruct`) and other data representation generators. - - `embedders.py`: Manages the connection to the **LanceDB** vector database and handles the indexing of vector embeddings. -- `rag_system/retrieval/`: Contains modules for retrieving and ranking documents. - - `retrievers.py`: Implements the logic for searching the vector database to find relevant text and image chunks. - - `reranker.py`: Contains the `QwenReranker` class, which re-ranks the retrieved documents for improved relevance. -- `rag_system/agent/`: Contains the `Agent` loop that interacts with the user and the RAG pipelines. -- `rag_system/utils/`: Contains utility clients, such as the `OllamaClient` for interacting with the Ollama server. - -### Data Flow: - -1. **Indexing**: - - The `MultimodalProcessor` reads a PDF and splits it into pages. - - For each page, it extracts the raw text and a full-page image. - - The `QwenEmbedder` generates a vector embedding for the text. - - The `LocalVisionModel` (using `colqwen2-v1.0`) generates a vector embedding for the image. - - The `VectorIndexer` stores these embeddings in separate tables within a **LanceDB** database. -2. **Retrieval**: - - A user submits a query to the `Agent`. - - The `RetrievalPipeline`'s `MultiVectorRetriever` searches both the text and image tables in LanceDB for relevant chunks. - - The retrieved documents are passed to the `QwenReranker`, which re-orders them based on relevance to the query. - - The top-ranked documents (containing both text and image references) are passed to the Vision Language Model (`qwen-vl`). - - The VLM analyzes the text and images to extract key facts. - - A final text generation model (`llama3`) synthesizes these facts into a coherent, human-readable answer. +### Key modules + +- `main.py` โ€” the MASTER configuration (`OLLAMA_CONFIG`, `WATSONX_CONFIG`, + `EXTERNAL_MODELS`, `PIPELINE_CONFIGS`) plus a thin argparse CLI. It builds nothing + itself; the CLI delegates to the factory. +- `factory.py` โ€” the single factory: `get_agent(mode)`, `get_indexing_pipeline(mode)` + and `get_pipeline_config(mode)` (which returns a deep copy so per-request overrides + cannot mutate the master config). +- `api_server.py` โ€” the HTTP API on port 8001 (`/health`, `/models`, `/chat`, + `/chat/stream`, `/index`), built on the standard library's `http.server`. +- `ingestion/` โ€” everything that turns a file into chunks. + - `document_converter.py`: Docling converters (no-OCR, OCR, general) and OCR-engine + selection. + - `docling_chunker.py`: token-budgeted chunker that walks the DoclingDocument tree, + keeping tables atomic and recording the heading path on every chunk. + - `chunking.py`: `MarkdownRecursiveChunker`, the heading/paragraph recursive + splitter used as the `legacy` chunker and as the Docling chunker's fallback. +- `indexing/` โ€” everything that turns chunks into an index. + - `representations.py`: `QwenEmbedder` (local Hugging Face), `OllamaEmbedder`, + `EmbeddingGenerator` and the `select_embedder()` dispatcher. + - `embedders.py`: `LanceDBManager` and `VectorIndexer` (schema, incremental append, + vector-width guard). + - `contextualizer.py`: `ContextualEnricher`, which prepends an LLM summary of the + surrounding window to each chunk before embedding. + - `latechunk.py`: `LateChunkEncoder`. + - `overview_builder.py`: per-document overviews used by the triage router. +- `retrieval/` โ€” `retrievers.py` (`MultiVectorRetriever`) and + `query_transformer.py` (`QueryDecomposer`). +- `rerankers/` โ€” `reranker.py` (`CrossEncoderReranker`, the non-`rerankers`-library + fallback) and `sentence_pruner.py` (Provence sentence-level pruning). +- `pipelines/` โ€” `indexing_pipeline.py` and `retrieval_pipeline.py`. +- `agent/` โ€” `loop.py` (`Agent`: triage, decomposition, orchestration, semantic cache, + per-session history) and `verifier.py` (`Verifier`). +- `utils/` โ€” `ollama_client.py`, `watsonx_client.py`, `batch_processor.py`, + `logging_utils.py`. + +### Indexing data flow + +1. `DocumentConverter.convert_to_markdown()` converts the file to Markdown. PDFs are + probed with PyMuPDF for an existing text layer; only text-layer-less PDFs take the + OCR converter. +2. `DoclingChunker` (default) or `MarkdownRecursiveChunker` splits the document into + chunks with a token budget of `chunking.chunk_size`. +3. `OverviewBuilder` writes a one-paragraph overview per document to + `index_store/overviews/*.jsonl` (used later for query routing). +4. If `contextual_enricher.enabled`, `ContextualEnricher` asks the enrichment model for a + short summary of each chunk's neighbourhood and prepends it to the chunk text. The + untouched text is kept in `metadata.original_text`. +5. `EmbeddingGenerator` embeds the chunks in batches and `VectorIndexer` writes them to + the LanceDB table, then creates the native FTS index on the `text` column. +6. If late chunking is enabled, `LateChunkEncoder` re-encodes each document as a whole and + writes per-chunk vectors to `<table>_lc`. + +There is no knowledge-graph step: `graph_extractor.py`, `GraphRetriever`, +`GraphQueryTranslator` and the `retrieval.graph` / `graph_strategy` config blocks were +removed on 2026-08-09 (roadmap item 2.5). The path was unreachable, and the evidence is +against it โ€” GraphRAG loses on single-hop retrieval, its multi-hop gains are contested, +and it costs 41โ€“57x at indexing and up to ~377x in query tokens +(`Documentation/research/academic-evidence-2026.md` ยง6). + +### Retrieval data flow + +1. `Agent.run()` routes the query: the overview router first, then a + "history exists โ†’ treat as follow-up" short circuit, then an LLM triage fallback that + picks `rag_query` or `direct_answer`. `force_rag=True` skips triage + entirely and pins `rag_query`. +2. Non-`direct_answer` queries are checked against the in-process semantic cache + (cosine similarity โ‰ฅ `semantic_cache_threshold`, scoped per session by default). +3. If query decomposition is on, `QueryDecomposer` splits the query. With the shipped + default (`compose_from_sub_answers: false`, `pooled_first_stage: true`) the first + stage runs per sub-query, the candidates are pooled and de-duplicated, then get one + rerank pass and one synthesis; with `compose_from_sub_answers: true` each sub-query + gets its own full retrieval in parallel, up to 3 workers, and the answers are composed. +4. `RetrievalPipeline.run()` retrieves via `MultiVectorRetriever.retrieve()` in the + configured mode, optionally also querying the late-chunk table and merging neighbours. +4b. If `retrieval.retry.enabled` and the first pass found weak evidence โ€” measured as the + *contrast* between the top candidate and the background of the rest, not the raw top + similarity โ€” the query is reformulated once on the enrichment model and steps 4-5 run + again, keeping whichever result set scores better (roadmap 2.1). +5. The reranker (if enabled) reorders the candidates and keeps `reranker.top_k`. When + sub-queries are present each candidate is scored against all of them and the scores + combined with `query_decomposition.rerank_aggregate`. +6. Context expansion pulls ยฑ`context_window_size` neighbouring chunks from LanceDB. +7. Provence pruning (off unless requested) drops irrelevant sentences from each chunk. +8. The generation model synthesises the answer from the surviving chunks, streaming + tokens to the caller when an event callback is supplied. +9. If verification is enabled and there are source documents, `Verifier` grades the + answer and appends ` [Confidence: N%]` (plus a low-confidence warning) to the answer + string. ## 3. Models -This system relies on a suite of powerful, open-source models. +The defaults live in `main.py` and every one of them except the Provence pruner is +overridable with an environment variable. -| Component | Model | Framework | Purpose | -| --------------------- | ----------------------------------- | -------------- | ------------------------------------------- | -| **Image Embedding** | `vidore/colqwen2-v1.0` | `colpali` | Generates vector embeddings from images. | -| **Text Embedding** | `Qwen/Qwen2-7B-instruct` | `transformers` | Generates vector embeddings from text. | -| **Reranker** | `Qwen/Qwen-reranker` | `transformers` | Re-ranks retrieved documents for relevance. | -| **Vision Language Model** | `qwen2.5vl:7b` | `Ollama` | Extracts facts from text and images. | -| **Text Generation** | `llama3` | `Ollama` | Synthesizes the final answer. | +| Role | Default | Runtime | Env override | +| ---- | ------- | ------- | ------------ | +| **Generation** (answers, sub-answer composition) | `qwen3.5:9b` | Ollama | `GENERATION_MODEL` | +| **Enrichment / utility** (routing, triage, decomposition, contextual enrichment, overviews, verification) | `qwen3.5:4b` | Ollama | `ENRICHMENT_MODEL` | +| **Embedding** | `microsoft/harrier-oss-v1-0.6b` (MIT, 1024-dim) | `transformers`, in-process | `EMBEDDING_MODEL` | +| **Reranker** (on by default) | `Qwen/Qwen3-Reranker-4B` | own yes/no-logit scorer, loaded lazily | `RERANKER_MODEL` | +| **Sentence pruning** (optional) | `naver/provence-reranker-debertav3-v1` | `transformers`, in-process | โ€” (hardcoded in `rerankers/sentence_pruner.py`) | + +Documented alternatives: + +- Generation: `qwen3.6:27b` (high-end, ~17 GB) or `qwen3.5:4b` (light). +- Enrichment: `qwen3.5:2b` (light). +- Embedding: `Qwen/Qwen3-Embedding-4B` (2560-dim, 32K context; for multilingual or + long-context corpora), `Qwen/Qwen3-Embedding-0.6B` (1024-dim). +- Reranker: `BAAI/bge-reranker-v2-m3` (low latency; only pays off with a weaker + embedder than the default โ€” see [`../eval/DECISIONS.md`](../eval/DECISIONS.md)), + `answerdotai/answerai-colbert-small-v1` (late interaction โ€” also set + `reranker.model_type` to `colbert`), `Qwen/Qwen3-Reranker-0.6B`. + +Notes: + +- **Embedding dimensions are never hardcoded.** `VectorIndexer.index()` reads the width + from the vectors the loaded model actually produced and builds the LanceDB schema from + it. Appending vectors of a different width to an existing table raises an explicit + error โ€” **changing the embedding model requires re-indexing every existing index.** +- **The width check is not enough on its own**, because two different models can share + it (harrier-oss-v1-0.6b and Qwen3-Embedding-0.6B are both 1024-dim). Every table + therefore records the embedding model that wrote it and whether its vectors are + L2-normalized; indexing into or querying a table with a different embedder raises + `EmbedderMismatchError` and names the model to rebuild with. +- **Vectors are L2-normalized at write and query time**, so LanceDB's default L2 + ordering is the cosine ordering both model cards specify. Tables written before this + existed carry no marker: they keep working with unnormalized vectors and log a + warning recommending a rebuild. +- `QwenEmbedder` truncates inputs at `min(tokenizer.model_max_length, 8192)` tokens. +- `select_embedder()` treats a name containing `/` as a Hugging Face repo id and anything + else as an Ollama tag served through `/api/embeddings`. +- If the reranker cannot be loaded, the pipeline logs a warning and continues **without** + reranking rather than failing the query. ## 4. Configuration -All system configurations are centralized in `main.py`. +All configuration lives in `main.py`. -- **`OLLAMA_CONFIG`**: Defines the models that will be run via the Ollama server. This includes the final text generation model and the Vision Language Model. -- **`PIPELINE_CONFIGS`**: Contains the configurations for both the `indexing` and `retrieval` pipelines. Here you can specify: - - The paths for the LanceDB database and source documents. - - The names of the tables to be used for text and image embeddings. - - The Hugging Face model names for the text embedder, vision model, and reranker. - - Parameters for the reranker and retrieval process (e.g., `top_k`, `retrieval_k`). +- **`LLM_BACKEND`** โ€” `ollama` (default) or `watsonx`. See + [`../WATSONX_README.md`](../WATSONX_README.md). +- **`OLLAMA_CONFIG`** โ€” `host`, `generation_model`, `enrichment_model`. The generation + model writes user-facing answers; the enrichment model does all the utility work + (routing, triage, decomposition, contextual enrichment, document overviews, + verification). +- **`WATSONX_CONFIG`** โ€” credentials, URL and the two granite model ids. It has the same + `generation_model` / `enrichment_model` keys so it is a drop-in replacement. +- **`EXTERNAL_MODELS`** โ€” `embedding_model` and `reranker_model`, the two Hugging Face + models loaded in-process. +- **`PIPELINE_CONFIGS`** โ€” exactly two profiles, `default` and `fast`. Selected with + `--mode` on the CLI or the `RAG_CONFIG_MODE` environment variable for the API server. -To change a model, simply update the corresponding model name in this configuration file. +| | `default` | `fast` | +| --- | --- | --- | +| `retrieval.search_type` | `hybrid` | `vector_only` | +| `retrieval.latechunk.enabled` | `true` | `false` | +| `reranker.enabled` | `false` (top 10 when switched on) | `false` | +| `query_decomposition.enabled` | `true` | `false` | +| `verification.enabled` | `true` | `false` | +| `contextual_enricher.enabled` | `true` (window 1) | `false` | +| `retrieval_k` | 20 | 10 | +| `indexing` batch sizes | 50 embed / 10 enrich | 100 embed / 50 enrich | -## 5. Usage +Both profiles store vectors under `./lancedb` in the table `text_pages_v4`, use +`semantic_cache_threshold: 0.98` and `cache_scope: "session"`. -To run the system, you first need to ensure the required models are available. +A per-key table (including which keys the API can override per request) is in +[`DOCUMENTATION.md`](DOCUMENTATION.md#4-configuration-reference). -### Prerequisites: +## 5. Usage + +### Prerequisites -1. **Install Dependencies**: +1. **Python 3.10+** (3.11 recommended) and dependencies, installed from the repository + root: ```bash pip install -r requirements.txt ``` -2. **Download Ollama Models**: + The root `requirements.txt` is the complete local-run set. `rag_system/requirements.txt` + is a partial list of the heavy ML dependencies that additionally pins + `ibm-watsonx-ai` and the macOS-only `ocrmac`. The Docker images install + `requirements-docker.txt` instead. + +2. **Ollama models**: ```bash - ollama pull llama3 - ollama pull qwen2.5vl:7b + ollama pull qwen3.5:9b + ollama pull qwen3.5:4b ``` -3. **Hugging Face Models**: The `transformers` and `colpali` libraries will automatically download the required models the first time they are used. Ensure you have a stable internet connection. -### Running the System: +3. **Hugging Face models** โ€” the embedder, reranker and (if used) Provence weights are + downloaded on first use by `transformers`. Set `HF_TOKEN` if you point at a gated + repository. -1. **Execute the Main Script**: - ```bash - python rag_system/main.py - ``` -2. **Indexing**: The script will first run the indexing pipeline, processing any documents in the `rag_system/documents` directory and storing their embeddings in LanceDB. -3. **Querying**: Once indexing is complete, the RAG agent will be ready. You can ask questions about the documents you have indexed. - ``` - > What was the revenue growth in Q3? - ``` -4. **Exit**: To stop the agent, type `quit`. +### Command line + +Run as a module from the repository root โ€” `python rag_system/main.py` will not work +because the package needs to be importable: + +```bash +# Index one file or a whole directory (walks *.pdf, *.docx, *.html, *.htm, *.md, *.txt) +python -m rag_system.main index ./shared_uploads --mode default + +# Ask a single question and print the JSON result +python -m rag_system.main chat "What was the revenue growth in Q3?" + +# Start the HTTP API (equivalent to python -m rag_system.api_server) +python -m rag_system.main api --port 8001 +``` + +`--mode` accepts `default` or `fast`. There is no interactive REPL. + +### Programmatic + +```python +from rag_system.factory import get_agent, get_indexing_pipeline + +get_indexing_pipeline("default").run(["/abs/path/to/document.pdf"]) + +agent = get_agent("default") +result = agent.run("What does the contract say about termination?") +print(result["answer"]) +print(len(result["source_documents"])) +``` + +`Agent.run()` returns `{"answer": str, "source_documents": list[dict]}` โ€” there is no +top-level confidence field; the verifier appends its score to the answer string. + +### HTTP + +```bash +python -m rag_system.api_server # port 8001 +curl http://localhost:8001/health # {"status": "ok"} +curl -X POST http://localhost:8001/chat \ + -H 'Content-Type: application/json' \ + -d '{"query": "What is in these documents?"}' +``` + +Endpoints, accepted fields and defaults are documented in +[`DOCUMENTATION.md`](DOCUMENTATION.md#5-http-api). + +## 6. Where this fits + +The RAG API is one of four processes. The frontend (`:3000`) talks to the backend gateway +(`:8000`), which forwards chat and indexing to this API (`:8001`), which in turn calls +Ollama (`:11434`). The frontend also streams directly from `:8001/chat/stream`. See the +repository [`README.md`](../README.md) and `Documentation/` for the system-level view. diff --git a/rag_system/agent/escalation.py b/rag_system/agent/escalation.py new file mode 100644 index 00000000..839733db --- /dev/null +++ b/rag_system/agent/escalation.py @@ -0,0 +1,259 @@ +"""Full-document escalation wiring (roadmap item 4.1). + +**Off by default.** ``retrieval.document_escalation.enabled`` must be set to +``True`` for anything in this module to run; until it has been measured on the +gold set it is a flag, not a behaviour. + +What it does +------------ +When candidate selection finishes โ€” *after* the evidence-sufficiency retry +(ยง5) has had its one attempt โ€” and the evidence signal is still below +threshold, the top-ranked chunk's whole document is reassembled in chunk order +and appended to the synthesis context as one clearly-delimited block. One +document, one time per user query, capped at a token budget. The chunk +citations are untouched: escalation adds reading material, it does not add or +reorder sources. + +Why it is a subclass +-------------------- +The decision point sits *between* candidate selection and synthesis, and both +of those live inside ``RetrievalPipeline.run()``. Rather than restructure +``rag_system/pipelines/retrieval_pipeline.py`` (owned by another workstream this +wave), this subclass hooks the two methods ``run()`` already calls in sequence: + +* ``retrieve_candidates()`` โ€” read the final, post-retry evidence signal and + remember the top-ranked chunk; +* ``_synthesize_final_answer()`` โ€” append the document to the facts string + before the base implementation builds its prompt. + +That keeps escalation to a **single** generation pass โ€” the alternative, +re-synthesising after ``run()`` returns, would pay for two answers and stream +tokens twice. The handoff between the two hooks is thread-local, so the agent's +parallel sub-query fan-out cannot cross-wire one sub-query's document into +another's synthesis. + +If a future refactor renames either hook, escalation stops firing and logs a +warning; it never breaks retrieval. +""" + +from __future__ import annotations + +import threading +from typing import Any, Dict, List, Optional + +from rag_system.pipelines.retrieval_pipeline import RetrievalPipeline +from rag_system.retrieval.document_fetch import fetch_document, format_escalation_block + +# Defaults live here, not in rag_system/main.py, because the profiles are not +# editable this wave. Every read goes through `config.get(...)` with these +# fallbacks, so adding the block to a profile later changes behaviour without +# touching this file. The keys the gate should add are listed in +# eval/decisions/phase4-escalation-tokens.md. +DEFAULT_DOCUMENT_ESCALATION: Dict[str, Any] = { + "enabled": False, + "max_documents": 1, + "token_budget": 6000, +} + +# Fallback trigger threshold, used only when neither the escalation block nor +# the retry block names one. Same value the shipped retry calibrated to. +_FALLBACK_MIN_EVIDENCE = 0.12 + + +class _EscalationBudget: + """How many documents this user query is still allowed to escalate.""" + + def __init__(self, max_documents: int) -> None: + self._lock = threading.Lock() + self._remaining = max(0, int(max_documents)) + self.events: List[Dict[str, Any]] = [] + + def take(self) -> bool: + with self._lock: + if self._remaining <= 0: + return False + self._remaining -= 1 + return True + + def note(self, payload: Dict[str, Any]) -> None: + with self._lock: + self.events.append(payload) + + +class EscalatingRetrievalPipeline(RetrievalPipeline): + """``RetrievalPipeline`` plus roadmap 4.1. Inert unless the flag is on.""" + + def __init__(self, *args, **kwargs) -> None: + super().__init__(*args, **kwargs) + self._escalation_local = threading.local() + self._escalation_budget: Optional[_EscalationBudget] = None + if not hasattr(RetrievalPipeline, "retrieve_candidates"): + print( + "โš ๏ธ RetrievalPipeline has no retrieve_candidates(); full-document " + "escalation (roadmap 4.1) is inert on this build." + ) + + # -------------------------------------------------------------- config + + def escalation_config(self) -> Dict[str, Any]: + """The ``document_escalation`` block, merged across both containers. + + Mirrors how ``_retry_config`` resolves ``retrieval.retry`` versus a + runtime ``retrievers.retry`` override. + """ + merged = dict(DEFAULT_DOCUMENT_ESCALATION) + for container_key in ("retrieval", "retrievers"): + block = (self.config.get(container_key) or {}).get("document_escalation") + if isinstance(block, dict): + merged.update(block) + return merged + + def _escalation_threshold(self, cfg: Dict[str, Any]) -> float: + """Below this evidence score, escalate. + + Defaults to the retry's own threshold: escalation is what happens when + the retry has already run and the evidence is *still* weak, so the two + should be judged against the same bar unless told otherwise. + """ + explicit = cfg.get("min_evidence") + if explicit is not None: + try: + return float(explicit) + except (TypeError, ValueError): + pass + try: + retry_cfg = self._retry_config() + except Exception: + retry_cfg = {} + for key in ("min_top_score", "min_rerank_score"): + value = retry_cfg.get(key) + if value is not None: + try: + return float(value) + except (TypeError, ValueError): + continue + return _FALLBACK_MIN_EVIDENCE + + # ------------------------------------------------------- request scope + + def begin_escalation_request(self) -> Optional[_EscalationBudget]: + """Open a per-user-query escalation budget. Returns ``None`` when off.""" + cfg = self.escalation_config() + if not cfg.get("enabled"): + self._escalation_budget = None + return None + self._escalation_budget = _EscalationBudget(cfg.get("max_documents", 1)) + return self._escalation_budget + + def end_escalation_request(self) -> Optional[_EscalationBudget]: + budget, self._escalation_budget = self._escalation_budget, None + self._escalation_local.pending = None + return budget + + # -------------------------------------------------------------- hooks + + def retrieve_candidates(self, query, table_name=None, sub_queries=None, + event_callback=None, *, filters=None): + # `filters` (roadmap 4.4) restores signature parity with the base + # class: keyword-only and passed straight through โ€” the base compiles + # it and opens its filter_scope around the retrieval. None keeps the + # thread-local scope opened by run(), so the default path is unchanged. + result = super().retrieve_candidates(query, table_name, sub_queries, + event_callback, filters=filters) + self._escalation_local.pending = None + if self._escalation_budget is None: + return result + try: + self._escalation_local.pending = self._plan_escalation(result, table_name) + except Exception as e: # never let observability break retrieval + print(f"โš ๏ธ Escalation planning failed: {e}") + self._escalation_local.pending = None + return result + + def _synthesize_final_answer(self, query: str, facts: str, *, event_callback=None) -> str: + plan = getattr(self._escalation_local, "pending", None) + self._escalation_local.pending = None + if plan is not None: + block = self._materialise(plan, event_callback) + if block: + facts = f"{facts}\n\n{block}" + return super()._synthesize_final_answer(query, facts, event_callback=event_callback) + + # ------------------------------------------------------------ internals + + def _plan_escalation(self, candidates: Dict[str, Any], + table_name: Optional[str]) -> Optional[Dict[str, Any]]: + """Decide whether the *final* evidence still warrants a deep read.""" + documents = candidates.get("documents") or [] + first_stage = candidates.get("first_stage") or [] + if not documents and not first_stage: + return None + + # Same preference order the retry uses: a calibrated reranker + # probability when there is one, else the dense contrast score. + score = self._rerank_evidence_score(documents) + signal = "rerank" + if score is None: + score = self._dense_evidence_score(first_stage) + signal = "dense_contrast" + if score is None: + # fts_only, or a legacy unnormalized table: no signal, no escalation + # โ€” exactly the rule the retry follows. + return None + + cfg = self.escalation_config() + threshold = self._escalation_threshold(cfg) + if score >= threshold: + return None + + top = (documents or first_stage)[0] + document_id = top.get("document_id") or (top.get("metadata") or {}).get("document_id") + if not document_id: + return None + + return { + "document_id": document_id, + "table_name": table_name or self.storage_config.get("text_table_name"), + "token_budget": int(cfg.get("token_budget", 6000) or 0), + "signal": signal, + "score": round(float(score), 4), + "threshold": float(threshold), + } + + def _materialise(self, plan: Dict[str, Any], event_callback) -> Optional[str]: + budget = self._escalation_budget + if budget is None or not budget.take(): + return None + try: + document = fetch_document( + self._get_db_manager(), + plan["table_name"], + plan["document_id"], + token_budget=plan["token_budget"], + ) + except Exception as e: + print(f"โš ๏ธ Full-document escalation failed: {e}") + return None + if document is None: + return None + + payload = document.as_event_payload() + payload.update({ + "signal": plan["signal"], + "score": plan["score"], + "threshold": plan["threshold"], + "token_budget": plan["token_budget"], + }) + budget.note(payload) + print( + f"\n๐Ÿ“„ Full-document escalation: {payload['document_name']} " + f"({payload['chunks_used']}/{payload['chunks_total']} chunks, " + f"~{payload['approx_tokens']} tokens, {plan['signal']}=" + f"{plan['score']} < {plan['threshold']})" + ) + if event_callback: + try: + event_callback("document_escalation", payload) + except Exception: + pass + return format_escalation_block(document) diff --git a/rag_system/agent/loop.py b/rag_system/agent/loop.py index b7da704d..9b3f8b33 100644 --- a/rag_system/agent/loop.py +++ b/rag_system/agent/loop.py @@ -1,14 +1,56 @@ from typing import Dict, Any, Optional +import contextlib +import contextvars +import copy import json +import re import time, asyncio, os import numpy as np import concurrent.futures from cachetools import TTLCache, LRUCache -from rag_system.utils.ollama_client import OllamaClient -from rag_system.pipelines.retrieval_pipeline import RetrievalPipeline +from rag_system.utils.ollama_client import ( + OllamaClient, + TokenUsageTracker, + token_stage, + track_token_usage, +) +from rag_system.agent.escalation import EscalatingRetrievalPipeline from rag_system.agent.verifier import Verifier -from rag_system.retrieval.query_transformer import QueryDecomposer, GraphQueryTranslator -from rag_system.retrieval.retrievers import GraphRetriever +from rag_system.retrieval.filters import compile_filters +from rag_system.retrieval.query_transformer import QueryDecomposer + +# Per-session history cap, in turns. The LRUCache on chat_histories bounds +# session COUNT; nothing else bounded turns within a session. +_MAX_HISTORY_TURNS = 40 + +# The verifier's answer suffixes: " [Confidence: 85%]" plus the optional +# low-confidence warning. Per-response UX only โ€” they must not be stored in +# history, where they would leak into later prompts (and into the +# decomposer's last_assistant_answer). +_ANSWER_TAGS_RE = re.compile( + r"\s*\[Confidence: \d+%\](\s*\[Warning: Low confidence\. Groundedness: [^\]]*\])?\s*$" +) + + +def _strip_answer_tags(answer: str) -> str: + """Return *answer* without the verifier's trailing confidence/warning tags.""" + return _ANSWER_TAGS_RE.sub("", answer) if answer else answer + + +def _in_copied_context(fn): + """Run *fn* in a copy of the caller's context, for ThreadPoolExecutor. + + ``submit()`` does not propagate context variables to the worker thread, so + without this the per-query token tracker (roadmap 4.5) would silently miss + every LLM call made by a parallel sub-query. One fresh copy per submitted + task โ€” a ``Context`` cannot be entered twice concurrently. + """ + ctx = contextvars.copy_context() + + def _runner(*args, **kwargs): + return ctx.run(fn, *args, **kwargs) + + return _runner class Agent: """ @@ -18,40 +60,52 @@ def __init__(self, pipeline_configs: Dict[str, Dict], llm_client: OllamaClient, self.pipeline_configs = pipeline_configs self.llm_client = llm_client self.ollama_config = ollama_config - - gen_model = self.ollama_config["generation_model"] - - # Initialize the single, persistent retrieval pipeline for this agent - self.retrieval_pipeline = RetrievalPipeline(pipeline_configs, self.llm_client, self.ollama_config) - - self.verifier = Verifier(llm_client, gen_model) - self.query_decomposer = QueryDecomposer(llm_client, gen_model) - + + # Utility work (routing, triage, decomposition, verification) runs on the + # small enrichment model; only user-facing answers use the generation model. + utility_model = self._utility_model() + + # Initialize the single, persistent retrieval pipeline for this agent. + # EscalatingRetrievalPipeline is a plain RetrievalPipeline with roadmap + # 4.1's full-document escalation hooked in; the hooks are inert unless + # `retrieval.document_escalation.enabled` is set, which it is not by + # default. See rag_system/agent/escalation.py. + self.retrieval_pipeline = EscalatingRetrievalPipeline( + pipeline_configs, self.llm_client, self.ollama_config + ) + + # `verification.model` (or the VERIFIER_MODEL env var, read inside + # Verifier) swaps the LLM-prompt verifier for a local NLI/verifier model. + # Unset by default โ€” the LLM prompt is what ships (roadmap 2.4). + verification_config = self.pipeline_configs.get("verification", {}) or {} + self.verifier = Verifier( + llm_client, + utility_model, + model_name=verification_config.get("model"), + threshold=float(verification_config.get("threshold", 0.5)), + ) + self.query_decomposer = QueryDecomposer(llm_client, utility_model) + # ๐Ÿš€ OPTIMIZED: TTL cache now stores embeddings for semantic matching - self._cache_max_size = 100 # fallback size limit for manual eviction helper - self._query_cache: TTLCache = TTLCache(maxsize=self._cache_max_size, ttl=300) + self._query_cache: TTLCache = TTLCache(maxsize=100, ttl=300) self.semantic_cache_threshold = self.pipeline_configs.get("semantic_cache_threshold", 0.98) - # If set to "session", semantic-cache hits will be restricted to the same chat session. - # Otherwise (default "global") answers can be reused across sessions. - self.cache_scope = self.pipeline_configs.get("cache_scope", "global") # 'global' or 'session' - + # If set to "global", semantic-cache hits are reused across chat sessions. + # The default keeps answers (and therefore document content) inside one session. + self.cache_scope = self.pipeline_configs.get("cache_scope", "session") # 'global' or 'session' + # ๐Ÿš€ NEW: In-memory store for conversational history per session self.chat_histories: LRUCache = LRUCache(maxsize=100) # Stores history for 100 recent sessions - graph_config = self.pipeline_configs.get("graph_strategy", {}) - if graph_config.get("enabled"): - self.graph_query_translator = GraphQueryTranslator(llm_client, gen_model) - self.graph_retriever = GraphRetriever(graph_config["graph_path"]) - print("Agent initialized with live GraphRAG capabilities.") - else: - print("Agent initialized (GraphRAG disabled).") - # ---- Load document overviews for fast routing ---- self._global_overview_path = os.path.join("index_store", "overviews", "overviews.jsonl") self.doc_overviews: list[str] = [] self._current_overview_session: str | None = None # cache key to avoid rereading on every query self._load_overviews(self._global_overview_path) + def _utility_model(self) -> str: + """Model used for routing, triage, decomposition and verification.""" + return self.ollama_config.get("enrichment_model") or self.ollama_config["generation_model"] + def _load_overviews(self, path: str): """Helper to load overviews from a .jsonl file into self.doc_overviews.""" import json, os @@ -122,7 +176,8 @@ def _cosine_similarity(self, v1: np.ndarray, v2: np.ndarray) -> float: return dot_product / (norm_v1 * norm_v2) - def _find_in_semantic_cache(self, query_embedding: np.ndarray, session_id: Optional[str] = None) -> Optional[Dict[str, Any]]: + def _find_in_semantic_cache(self, query_embedding: np.ndarray, session_id: Optional[str] = None, + filter_signature: Optional[str] = None) -> Optional[Dict[str, Any]]: """Finds a semantically similar query in the cache.""" if not self._query_cache or query_embedding is None: return None @@ -133,9 +188,16 @@ def _find_in_semantic_cache(self, query_embedding: np.ndarray, session_id: Optio continue # Respect cache scoping: if scope is session-level, skip results from other sessions - if self.cache_scope == "session" and session_id is not None: - if cached_item.get("session_id") != session_id: - continue + if self.cache_scope != "global" and cached_item.get("session_id") != session_id: + continue + + # A metadata filter (item 4.4) is part of the question, not a view + # over the answer: "what does the NDA say" and "what do all ten + # documents say" have near-identical embeddings and different right + # answers. Both sides are None on the unfiltered path, so this is a + # no-op unless someone actually filtered. + if cached_item.get("filters") != filter_signature: + continue try: similarity = self._cosine_similarity(query_embedding, cached_embedding) @@ -168,14 +230,26 @@ def _format_query_with_history(self, query: str, history: list) -> str: return prompt # ---------------- Asynchronous triage using Ollama ---------------- + @staticmethod + def _normalize_triage(decision: str) -> str: + """Collapse a triage label to one of the two categories that still exist. + + ``graph_query`` was a third outcome until the graph module was removed + (roadmap 2.5, 2026-08-09). Small utility models still emit it from time to + time โ€” they have seen the label in their training data โ€” so anything that + is not an explicit ``direct_answer`` is answered from the documents. + """ + return "direct_answer" if (decision or "").strip() == "direct_answer" else "rag_query" + async def _triage_query_async(self, query: str, history: list) -> str: - + print(f"๐Ÿ” ROUTING DEBUG: Starting triage for query: '{query[:100]}...'") # 1๏ธโƒฃ Fast routing using precomputed overviews (if available) print(f"๐Ÿ“– ROUTING DEBUG: Attempting overview-based routing...") routed = self._route_via_overviews(query) if routed: + routed = self._normalize_triage(routed) print(f"โœ… ROUTING DEBUG: Overview routing decided: '{routed}'") return routed else: @@ -198,8 +272,6 @@ async def _triage_query_async(self, query: str, history: list) -> str: 2. "direct_answer" โ€“ General knowledge questions, greetings, or queries unrelated to uploaded documents. Examples: "Who are the CEOs of Tesla and Amazon?", "What is the capital of France?", "Hello", "Explain quantum physics" -3. "graph_query" โ€“ Specific factual relations for knowledge-graph lookup (currently limited use) - IMPORTANT: For general world knowledge about well-known companies, people, or facts NOT related to uploaded documents, choose "direct_answer". User query: "{query}" @@ -207,123 +279,157 @@ async def _triage_query_async(self, query: str, history: list) -> str: Respond with JSON: {{"category": "<your_choice>"}} """ resp = self.llm_client.generate_completion( - model=self.ollama_config["generation_model"], prompt=prompt, format="json" + model=self._utility_model(), prompt=prompt, format="json", + options={"temperature": 0}, # deterministic routing (same pin the eval judge got in ebcc88b) ) try: data = json.loads(resp.get("response", "{}")) - decision = data.get("category", "rag_query") + decision = self._normalize_triage(data.get("category", "rag_query")) print(f"๐Ÿค– ROUTING DEBUG: LLM fallback triage decided: '{decision}'") return decision except json.JSONDecodeError: print(f"โŒ ROUTING DEBUG: LLM fallback triage JSON parsing failed, defaulting to 'rag_query'") return "rag_query" - def _run_graph_query(self, query: str, history: list) -> Dict[str, Any]: - contextual_query = self._format_query_with_history(query, history) - structured_query = self.graph_query_translator.translate(contextual_query) - if not structured_query.get("start_node"): - return self.retrieval_pipeline.run(contextual_query, window_size_override=0) - results = self.graph_retriever.retrieve(structured_query) - if not results: - return self.retrieval_pipeline.run(contextual_query, window_size_override=0) - answer = ", ".join([res['details']['node_id'] for res in results]) - return {"answer": f"From the knowledge graph: {answer}", "source_documents": results} - - def _get_cache_key(self, query: str, query_type: str) -> str: - """Generate a cache key for the query""" - # Simple cache key based on query and type - return f"{query_type}:{query.strip().lower()}" - - def _cache_result(self, cache_key: str, result: Dict[str, Any], session_id: Optional[str] = None): - """Cache a result with size limit""" - if len(self._query_cache) >= self._cache_max_size: - # Remove oldest entry (simple FIFO eviction) - oldest_key = next(iter(self._query_cache)) - del self._query_cache[oldest_key] - - self._query_cache[cache_key] = { - 'result': result, - 'timestamp': time.time(), - 'session_id': session_id - } - # ---------------- Public sync API (kept for backwards compatibility) -------------- - def run(self, query: str, table_name: str = None, session_id: str = None, compose_sub_answers: Optional[bool] = None, query_decompose: Optional[bool] = None, ai_rerank: Optional[bool] = None, context_expand: Optional[bool] = None, verify: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, search_type: Optional[str] = None, dense_weight: Optional[float] = None, max_retries: int = 1, event_callback: Optional[callable] = None) -> Dict[str, Any]: + def run(self, query: str, table_name: str = None, session_id: str = None, compose_sub_answers: Optional[bool] = None, query_decompose: Optional[bool] = None, ai_rerank: Optional[bool] = None, context_expand: Optional[bool] = None, verify: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, retrieval_mode: Optional[str] = None, force_rag: bool = False, event_callback: Optional[callable] = None, *, filters: Any = None) -> Dict[str, Any]: """Synchronous helper. If *event_callback* is supplied, important milestones will be forwarded to that callable as event_callback(phase:str, payload:Any) + + *filters* (roadmap item 4.4) is a metadata filter object; it is + keyword-only and defaults to None, in which case nothing about this + call changes. + """ + return asyncio.run(self._run_async(query, table_name, session_id, compose_sub_answers, query_decompose, ai_rerank, context_expand, verify, retrieval_k, context_window_size, reranker_top_k, retrieval_mode, force_rag, event_callback, filters=filters)) + + # ---------------- Per-query bookkeeping wrapper ----------------------------------- + @contextlib.contextmanager + def _pipeline_config_overrides(self, *, ai_rerank: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, retrieval_mode: Optional[str] = None): + """Apply per-request retrieval overrides, then restore the shared config. + + The retrieval pipeline is a process-wide singleton, so writing these + straight into ``pipeline.config`` made one request's toggle sticky for + every later request. Snapshot the affected subtrees, apply, and restore + in ``finally`` โ€” the same pattern as ``_generation_model_override`` in + api_server.py. Snapshot/restore is safe because the RAG API server is + single-threaded by design; a concurrent server would need a lock. + """ + config = self.retrieval_pipeline.config + missing = object() + snapshot = {} + for key in ("reranker", "retrieval", "retrieval_k", "context_window_size"): + value = config.get(key, missing) + snapshot[key] = copy.deepcopy(value) if value is not missing else value + + if ai_rerank is not None: + rr_cfg = config.setdefault("reranker", {}) + rr_cfg["enabled"] = bool(ai_rerank) + if retrieval_k is not None: + config["retrieval_k"] = retrieval_k + print(f"๐Ÿ” Retrieval K set to: {retrieval_k}") + if context_window_size is not None: + config["context_window_size"] = context_window_size + print(f"๐Ÿ” Context window size set to: {context_window_size}") + if reranker_top_k is not None: + rr_cfg = config.setdefault("reranker", {}) + rr_cfg["top_k"] = reranker_top_k + print(f"๐Ÿ” Reranker top K set to: {reranker_top_k}") + if retrieval_mode is not None: + retrieval_cfg = config.setdefault("retrieval", {}) + retrieval_cfg["search_type"] = retrieval_mode + print(f"๐Ÿ” Retrieval mode set to: {retrieval_mode}") + + try: + yield + finally: + for key, value in snapshot.items(): + if value is missing: + config.pop(key, None) + else: + config[key] = value + + async def _run_async(self, query: str, table_name: str = None, session_id: str = None, compose_sub_answers: Optional[bool] = None, query_decompose: Optional[bool] = None, ai_rerank: Optional[bool] = None, context_expand: Optional[bool] = None, verify: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, retrieval_mode: Optional[str] = None, force_rag: bool = False, event_callback: Optional[callable] = None, *, filters: Any = None) -> Dict[str, Any]: + """Wraps one user query in its token-usage, config-override and escalation scopes. + + All three are per-*query*, not per-retrieval, which is why they live out + here rather than inside the pipeline: a decomposed query runs the + pipeline several times and must still report one token total, see one + set of runtime overrides and escalate at most one document + (roadmap 4.1 / 4.5). """ - return asyncio.run(self._run_async(query, table_name, session_id, compose_sub_answers, query_decompose, ai_rerank, context_expand, verify, retrieval_k, context_window_size, reranker_top_k, search_type, dense_weight, max_retries, event_callback)) + tracker = TokenUsageTracker() + pipeline = self.retrieval_pipeline + can_escalate = hasattr(pipeline, "begin_escalation_request") + + with track_token_usage(tracker): + if can_escalate: + pipeline.begin_escalation_request() + try: + with self._pipeline_config_overrides( + ai_rerank=ai_rerank, + retrieval_k=retrieval_k, + context_window_size=context_window_size, + reranker_top_k=reranker_top_k, + retrieval_mode=retrieval_mode, + ): + result = await self._run_async_inner(query, table_name, session_id, compose_sub_answers, query_decompose, ai_rerank, context_expand, verify, retrieval_k, context_window_size, reranker_top_k, retrieval_mode, force_rag, event_callback, filters=filters) + finally: + budget = pipeline.end_escalation_request() if can_escalate else None + + if not isinstance(result, dict): + return result + + # Shallow-copied so writing the usage of *this* run cannot mutate an + # entry the semantic cache is holding on to. + result = dict(result) + result["token_usage"] = tracker.as_dict() + if budget is not None and budget.events: + result["document_escalation"] = budget.events + return result # ---------------- Main async implementation -------------------------------------- - async def _run_async(self, query: str, table_name: str = None, session_id: str = None, compose_sub_answers: Optional[bool] = None, query_decompose: Optional[bool] = None, ai_rerank: Optional[bool] = None, context_expand: Optional[bool] = None, verify: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, search_type: Optional[str] = None, dense_weight: Optional[float] = None, max_retries: int = 1, event_callback: Optional[callable] = None) -> Dict[str, Any]: + async def _run_async_inner(self, query: str, table_name: str = None, session_id: str = None, compose_sub_answers: Optional[bool] = None, query_decompose: Optional[bool] = None, ai_rerank: Optional[bool] = None, context_expand: Optional[bool] = None, verify: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, retrieval_mode: Optional[str] = None, force_rag: bool = False, event_callback: Optional[callable] = None, *, filters: Any = None) -> Dict[str, Any]: start_time = time.time() - + + # Compiled once per user query (roadmap 4.4) and handed to every + # retrieval this query performs, including each parallel sub-query. + # Raises FilterError on an invalid filter โ€” /chat validates first and + # returns 400, so reaching here with a bad filter means a non-HTTP caller. + compiled_filters = compile_filters(filters) + filter_signature = compiled_filters.signature if compiled_filters else None + # Emit analyze event at the start if event_callback: event_callback("analyze", {"query": query}) # ๐Ÿš€ NEW: Get conversation history history = self.chat_histories.get(session_id, []) if session_id else [] - - # ๐Ÿ”„ Refresh overviews for this session if available - # if session_id and session_id != getattr(self, "_current_overview_session", None): - # candidate_path = os.path.join("index_store", "overviews", f"{session_id}.jsonl") - # if os.path.exists(candidate_path): - # self._load_overviews(candidate_path) - # self._current_overview_session = session_id - # else: - # # Fall back to global overviews if per-session file not found - # if self._current_overview_session != "GLOBAL": - # self._load_overviews(self._global_overview_path) - # self._current_overview_session = "GLOBAL" - - query_type = await self._triage_query_async(query, history) - print(f"๐ŸŽฏ ROUTING DEBUG: Final triage decision: '{query_type}'") + + if force_rag or compiled_filters is not None: + # A metadata filter is an instruction to search a named part of the + # corpus (roadmap 4.4). Letting triage answer it from the model's + # general knowledge instead would ignore the filter completely and + # look, from the outside, exactly like a filter that matched + # nothing โ€” so a filter skips triage the way force_rag does. + query_type = "rag_query" + reason = "force_rag" if force_rag else "metadata filter" + print(f"๐ŸŽฏ ROUTING DEBUG: {reason} set โ€“ triage skipped, using 'rag_query'") + else: + with token_stage("triage"): + query_type = await self._triage_query_async(query, history) + print(f"๐ŸŽฏ ROUTING DEBUG: Final triage decision: '{query_type}'") print(f"Agent Triage Decision: '{query_type}'") - + # Create a contextual query that includes history for most operations contextual_query = self._format_query_with_history(query, history) raw_query = query.strip() - - # --- Apply runtime AI reranker override (must happen before any retrieval calls) --- - if ai_rerank is not None: - rr_cfg = self.retrieval_pipeline.config.setdefault("reranker", {}) - rr_cfg["enabled"] = bool(ai_rerank) - if ai_rerank: - # Ensure the pipeline knows to use the external ColBERT reranker - rr_cfg.setdefault("type", "ai") - rr_cfg.setdefault("strategy", "rerankers-lib") - rr_cfg.setdefault( - "model_name", - # Falls back to ColBERT-small if the caller did not supply one - self.ollama_config.get("rerank_model", "answerai-colbert-small-v1"), - ) - - # --- Apply runtime retrieval configuration overrides --- - if retrieval_k is not None: - self.retrieval_pipeline.config["retrieval_k"] = retrieval_k - print(f"๐Ÿ” Retrieval K set to: {retrieval_k}") - - if context_window_size is not None: - self.retrieval_pipeline.config["context_window_size"] = context_window_size - print(f"๐Ÿ” Context window size set to: {context_window_size}") - - if reranker_top_k is not None: - rr_cfg = self.retrieval_pipeline.config.setdefault("reranker", {}) - rr_cfg["top_k"] = reranker_top_k - print(f"๐Ÿ” Reranker top K set to: {reranker_top_k}") - - if search_type is not None: - retrieval_cfg = self.retrieval_pipeline.config.setdefault("retrieval", {}) - retrieval_cfg["search_type"] = search_type - print(f"๐Ÿ” Search type set to: {search_type}") - - if dense_weight is not None: - dense_cfg = self.retrieval_pipeline.config.setdefault("retrieval", {}).setdefault("dense", {}) - dense_cfg["weight"] = dense_weight - print(f"๐Ÿ” Dense search weight set to: {dense_weight}") + + # Runtime retrieval overrides (ai_rerank, retrieval_k, context_window_size, + # reranker_top_k, retrieval_mode) are applied around this whole method by + # _run_async's _pipeline_config_overrides scope and rolled back there, so + # everything below already sees this request's config. query_embedding = None # ๐Ÿš€ OPTIMIZED: Semantic Cache Check @@ -338,13 +444,14 @@ async def _run_async(self, query: str, table_name: str = None, session_id: str = # Some embedders return a list โ€“ convert if necessary query_embedding = np.array(query_embedding_list[0]) - cached_result = self._find_in_semantic_cache(query_embedding, session_id) + cached_result = self._find_in_semantic_cache(query_embedding, session_id, + filter_signature) if cached_result: # Update history even on cache hit if session_id: - history.append({"query": query, "answer": cached_result.get('answer', 'Cached answer not found.')}) - self.chat_histories[session_id] = history + history.append({"query": query, "answer": _strip_answer_tags(cached_result.get('answer', 'Cached answer not found.'))}) + self.chat_histories[session_id] = history[-_MAX_HISTORY_TURNS:] return cached_result if query_type == "direct_answer": @@ -363,12 +470,15 @@ async def _run_stream(): answer_parts: list[str] = [] def _blocking_stream(): - for tok in self.llm_client.stream_completion( - model=self.ollama_config["generation_model"], prompt=prompt - ): - answer_parts.append(tok) - if event_callback: - event_callback("token", {"text": tok}) + with token_stage("synthesis"): + for tok in self.llm_client.stream_completion( + model=self.ollama_config["generation_model"], prompt=prompt, + enable_thinking=False, # thinking burns the window and can yield an empty answer + options={"temperature": 0}, # greedy decode, matches synthesis (arm C config) + ): + answer_parts.append(tok) + if event_callback: + event_callback("token", {"text": tok}) # Run the blocking generator in a thread so the event loop stays responsive await asyncio.to_thread(_blocking_stream) @@ -376,10 +486,6 @@ def _blocking_stream(): final_answer = await _run_stream() result = {"answer": final_answer, "source_documents": []} - - elif query_type == "graph_query" and hasattr(self, 'graph_retriever'): - print(f"โœ… ROUTING DEBUG: Executing GRAPH_QUERY path") - result = self._run_graph_query(query, history) # --- RAG Query Processing with Optional Query Decomposition --- else: # Default to rag_query @@ -394,26 +500,37 @@ def _blocking_stream(): # Use the raw user query (without conversation history) for decomposition to avoid leakage of prior context # Pass the last 5 conversation turns for context resolution within the decomposer recent_history = history[-5:] if history else [] - sub_queries = self.query_decomposer.decompose(raw_query, recent_history) + with token_stage("decomposition"): + sub_queries = self.query_decomposer.decompose( + raw_query, + recent_history, + max_sub_queries=query_decomp_config.get("max_sub_queries", 10), + resolve_only=bool(query_decomp_config.get("resolve_only", False)), + ) if event_callback: event_callback("decomposition", {"sub_queries": sub_queries}) print(f"Original query: '{query}' (Contextual: '{contextual_query}')") print(f"Decomposed into {len(sub_queries)} sub-queries: {sub_queries}") - - # Emit retrieval_started event before any retrievals - if event_callback: - event_callback("retrieval_started", {"count": len(sub_queries)}) - + + # retrieval_started is emitted per branch below, with the count + # that branch actually retrieves with โ€” 1 for the single and + # pooled paths, N for compose. One agent-level event per + # logical retrieval, so the frontend never sees contradictory + # counts (the pipeline emits its own retrieval_started too). # If decomposition produced only a single sub-query, skip the # parallel/composition machinery for efficiency. if len(sub_queries) == 1: print("--- Only one sub-query after decomposition; using direct retrieval path ---") - result = self.retrieval_pipeline.run( - sub_queries[0], - table_name, - 0 if context_expand is False else None, - event_callback=event_callback - ) + if event_callback: + event_callback("retrieval_started", {"count": 1}) + with token_stage("synthesis"): + result = self.retrieval_pipeline.run( + sub_queries[0], + table_name, + 0 if context_expand is False else None, + event_callback=event_callback, + filters=compiled_filters, + ) if event_callback: event_callback("single_query_result", result) # Emit retrieval_done and rerank_done for single sub-query @@ -426,59 +543,105 @@ def _blocking_stream(): if compose_sub_answers is not None: compose_from_sub_answers = compose_sub_answers - print(f"\n--- Processing {len(sub_queries)} sub-queries in parallel ---") - start_time_inner = time.time() - - # Shared containers - sub_answers = [] # For two-stage composition - all_source_docs = [] # For single-stage aggregation - citations_seen = set() - - # Emit rerank_started event before parallel retrievals (since each sub-query will rerank) - if event_callback: - event_callback("rerank_started", {"count": len(sub_queries)}) - - # Emit token chunks as soon as we receive them. The UI - # keeps answers separated by `index`, so interleaving is - # harmless and gives continuous feedback. - - def make_cb(idx: int): - def _cb(ev_type: str, payload): - if event_callback is None: - return - if ev_type == "token": - event_callback("sub_query_token", {"index": idx, "text": payload.get("text", ""), "question": sub_queries[idx]}) - else: - event_callback(ev_type, payload) - return _cb - - with concurrent.futures.ThreadPoolExecutor(max_workers=min(3, len(sub_queries))) as executor: - future_to_query = { - executor.submit( - self.retrieval_pipeline.run, - sub_query, + if not compose_from_sub_answers: + # ---- Roadmap item 2.2: decomposition applies at RERANK ---- + # The first stage runs ONCE, on the full original query. + # Fanning the *first stage* out over sub-queries dilutes + # it semantically (2026 MultiConIR/SSRB finding); the + # sub-queries earn their keep at the rerank stage, where + # every candidate is scored against every sub-query and + # the scores are aggregated with + # `query_decomposition.rerank_aggregate` ("max" or "mean"). + # With reranking off there is no rerank stage, so the + # sub-queries go unused and this is plain single-query + # retrieval โ€” which is the shipped default. + if query_decomp_config.get("pooled_first_stage"): + print("\n--- Pooled decomposition: per-sub-query retrieval, " + "single rerank + synthesis ---") + else: + print("\n--- Decomposition applied at rerank; first stage uses the full query ---") + if event_callback: + event_callback("retrieval_started", {"count": 1}) + with token_stage("synthesis"): + result = self.retrieval_pipeline.run( + contextual_query, table_name, 0 if context_expand is False else None, - make_cb(i), - ): (i, sub_query) - for i, sub_query in enumerate(sub_queries) - } + event_callback=event_callback, + sub_queries=sub_queries, + filters=compiled_filters, + ) + if event_callback: + event_callback("final_answer", result) + else: + # `compose_from_sub_answers` keeps per-sub-query *retrieval* + # on purpose, and is the only thing that does: it needs a + # separate answer per sub-question to compose from, which a + # single shared candidate set cannot produce. One full + # RetrievalPipeline.run() โ€” retrieval, rerank, synthesis โ€” + # per sub-query, in parallel. + print(f"\n--- Processing {len(sub_queries)} sub-queries in parallel ---") + start_time_inner = time.time() + + sub_answers = [] + all_source_docs = [] + citations_seen = set() + + # One retrieval_started for the whole fan-out: each + # sub-query runs a full retrieval below. + if event_callback: + event_callback("retrieval_started", {"count": len(sub_queries)}) - for future in concurrent.futures.as_completed(future_to_query): - i, sub_query = future_to_query[future] - try: - sub_result = future.result() - print(f"โœ… Sub-Query {i+1} completed: '{sub_query}'") - - if event_callback: - event_callback("sub_query_result", { - "index": i, - "query": sub_query, - "answer": sub_result.get("answer", ""), - "source_documents": sub_result.get("source_documents", []), - }) + # Emit rerank_started before the parallel retrievals (each sub-query reranks). + if event_callback: + event_callback("rerank_started", {"count": len(sub_queries)}) + + # Emit token chunks as soon as we receive them. The UI + # keeps answers separated by `index`, so interleaving is + # harmless and gives continuous feedback. + + def make_cb(idx: int): + def _cb(ev_type: str, payload): + if event_callback is None: + return + if ev_type == "token": + event_callback("sub_query_token", {"index": idx, "text": payload.get("text", ""), "question": sub_queries[idx]}) + else: + event_callback(ev_type, payload) + return _cb + + with concurrent.futures.ThreadPoolExecutor(max_workers=min(3, len(sub_queries))) as executor: + # Each task runs in its own copy of the current + # context so the per-query token tracker and the + # "synthesis" stage label reach the worker thread + # (submit() does not propagate context variables). + with token_stage("synthesis"): + future_to_query = { + executor.submit( + _in_copied_context(self.retrieval_pipeline.run), + sub_query, + table_name, + 0 if context_expand is False else None, + make_cb(i), + filters=compiled_filters, + ): (i, sub_query) + for i, sub_query in enumerate(sub_queries) + } + + for future in concurrent.futures.as_completed(future_to_query): + i, sub_query = future_to_query[future] + try: + sub_result = future.result() + print(f"โœ… Sub-Query {i+1} completed: '{sub_query}'") + + if event_callback: + event_callback("sub_query_result", { + "index": i, + "query": sub_query, + "answer": sub_result.get("answer", ""), + "source_documents": sub_result.get("source_documents", []), + }) - if compose_from_sub_answers: sub_answers.append({ "question": sub_query, "answer": sub_result.get("answer", "") @@ -488,24 +651,24 @@ def _cb(ev_type: str, payload): if doc['chunk_id'] not in citations_seen: all_source_docs.append(doc) citations_seen.add(doc['chunk_id']) - else: - # Aggregate unique docs (single-stage path) - for doc in sub_result.get('source_documents', []): - if doc['chunk_id'] not in citations_seen: - all_source_docs.append(doc) - citations_seen.add(doc['chunk_id']) - except Exception as e: - print(f"โŒ Sub-Query {i+1} failed: '{sub_query}' - {e}") + except Exception as e: + print(f"โŒ Sub-Query {i+1} failed: '{sub_query}' - {e}") - parallel_time = time.time() - start_time_inner - print(f"๐Ÿš€ Parallel processing completed in {parallel_time:.2f}s") + print(f"๐Ÿš€ Parallel processing completed in {time.time() - start_time_inner:.2f}s") - # Emit retrieval_done and rerank_done after all sub-queries are processed - if event_callback: - event_callback("retrieval_done", {"count": len(sub_queries)}) - event_callback("rerank_done", {"count": len(sub_queries)}) + # Every sub-query raised: composing from "[]" would dress + # total failure up as a 200. Fail the way the single-query + # path does (raise -> 500 / SSE error); a partial failure + # stays best-effort. + if not sub_answers: + raise RuntimeError( + f"All {len(sub_queries)} sub-queries failed for query: '{raw_query}'" + ) + + if event_callback: + event_callback("retrieval_done", {"count": len(sub_queries)}) + event_callback("rerank_done", {"count": len(sub_queries)}) - if compose_from_sub_answers: print("\n--- Composing final answer from sub-answers ---") compose_prompt = f""" You are an expert answer composer for a Retrieval-Augmented Generation (RAG) system. @@ -536,13 +699,20 @@ def _cb(ev_type: str, payload): # --- Stream composition answer token-by-token --- answer_parts: list[str] = [] - for tok in self.llm_client.stream_completion( - model=self.ollama_config["generation_model"], - prompt=compose_prompt, - ): - answer_parts.append(tok) - if event_callback: - event_callback("token", {"text": tok}) + def _blocking_compose_stream(): + with token_stage("synthesis"): + for tok in self.llm_client.stream_completion( + model=self.ollama_config["generation_model"], + prompt=compose_prompt, + enable_thinking=False, # thinking burns the window and can yield an empty answer + options={"temperature": 0}, # greedy decode, matches synthesis (arm C config) + ): + answer_parts.append(tok) + if event_callback: + event_callback("token", {"text": tok}) + + # Run the blocking generator in a thread so the event loop stays responsive + await asyncio.to_thread(_blocking_compose_stream) final_answer = "".join(answer_parts) or "Unable to generate an answer." @@ -552,47 +722,12 @@ def _cb(ev_type: str, payload): } if event_callback: event_callback("final_answer", result) - else: - print(f"\n--- Aggregated {len(all_source_docs)} unique documents from all sub-queries ---") - - if all_source_docs: - aggregated_context = "\n\n".join([doc['text'] for doc in all_source_docs]) - final_answer = self.retrieval_pipeline._synthesize_final_answer(contextual_query, aggregated_context) - result = { - "answer": final_answer, - "source_documents": all_source_docs - } - if event_callback: - event_callback("final_answer", result) - else: - result = { - "answer": "I could not find relevant information to answer your question.", - "source_documents": [] - } - if event_callback: - event_callback("final_answer", result) else: # Standard retrieval (single-query) - retrieved_docs = (self.retrieval_pipeline.retriever.retrieve( - text_query=contextual_query, - table_name=table_name or self.retrieval_pipeline.storage_config["text_table_name"], - k=self.retrieval_pipeline.config.get("retrieval_k", 10), - ) if hasattr(self.retrieval_pipeline, "retriever") and self.retrieval_pipeline.retriever else []) - - print("\n=== DEBUG: Original retrieval order ===") - for i, d in enumerate(retrieved_docs[:10]): - snippet = (d.get('text','') or '')[:200].replace('\n',' ') - print(f"Orig[{i}] id={d.get('chunk_id')} dist={d.get('_distance','') or d.get('score','')} {snippet}") - - result = self.retrieval_pipeline.run(contextual_query, table_name, 0 if context_expand is False else None, event_callback=event_callback) - - # After run, result['source_documents'] is reranked list - reranked_docs = result.get('source_documents', []) - print("\n=== DEBUG: Reranked docs order ===") - for i, d in enumerate(reranked_docs[:10]): - snippet = (d.get('text','') or '')[:200].replace('\n',' ') - print(f"ReRank[{i}] id={d.get('chunk_id')} score={d.get('rerank_score','')} {snippet}") - + with token_stage("synthesis"): + result = self.retrieval_pipeline.run(contextual_query, table_name, 0 if context_expand is False else None, event_callback=event_callback, filters=compiled_filters) + + # Verification step (simplified for now) - Skip in fast mode verification_enabled = self.pipeline_configs.get("verification", {}).get("enabled", True) if verify is not None: @@ -600,7 +735,8 @@ def _cb(ev_type: str, payload): if verification_enabled and result.get("source_documents"): context_str = "\n".join([doc['text'] for doc in result['source_documents']]) - verification = await self.verifier.verify_async(contextual_query, context_str, result['answer']) + with token_stage("verification"): + verification = await self.verifier.verify_async(contextual_query, context_str, result['answer']) score = verification.confidence_score @@ -616,10 +752,12 @@ def _cb(ev_type: str, payload): else: print("๐Ÿš€ Skipping verification for speed or lack of sources") - # ๐Ÿš€ NEW: Update history + # ๐Ÿš€ NEW: Update history โ€” store the answer WITHOUT the verifier's + # confidence tags; they are per-response UX and would otherwise leak + # into later prompts (and the decomposer's last_assistant_answer). if session_id: - history.append({"query": query, "answer": result['answer']}) - self.chat_histories[session_id] = history + history.append({"query": query, "answer": _strip_answer_tags(result['answer'])}) + self.chat_histories[session_id] = history[-_MAX_HISTORY_TURNS:] # ๐Ÿš€ OPTIMIZED: Cache the result for future queries if query_type != "direct_answer" and query_embedding is not None: @@ -628,6 +766,7 @@ def _cb(ev_type: str, payload): "embedding": query_embedding, "result": result, "session_id": session_id, + "filters": filter_signature, } total_time = time.time() - start_time @@ -651,22 +790,24 @@ def _route_via_overviews(self, query: str) -> str | None: router_prompt = f"""Task: Route query to correct system. -Documents available: Invoices, DeepSeek-V3 research papers +DOCUMENT OVERVIEWS: +{overviews_block} Query: "{query}" Is this query asking about: A) Greetings/social: "Hi", "Hello", "Thanks", "What's up", "How are you" -B) General knowledge: "CEO of Tesla", "capital of France", "what is 2+2" -C) Document content: invoice amounts, DeepSeek-V3 details, companies mentioned +B) General knowledge unrelated to the documents above: "CEO of Tesla", "capital of France", "what is 2+2" +C) Anything covered by, or plausibly contained in, the documents above If A or B โ†’ {{"category": "direct_answer"}} If C โ†’ {{"category": "rag_query"}} Response:""" - + resp = self.llm_client.generate_completion( - model=self.ollama_config["generation_model"], prompt=router_prompt, format="json" + model=self._utility_model(), prompt=router_prompt, format="json", + options={"temperature": 0}, # deterministic routing (same pin the eval judge got in ebcc88b) ) try: raw_response = resp.get("response", "{}") diff --git a/rag_system/agent/verifier.py b/rag_system/agent/verifier.py index 89ae64c4..b6c900fb 100644 --- a/rag_system/agent/verifier.py +++ b/rag_system/agent/verifier.py @@ -1,27 +1,233 @@ +import asyncio import json +import os +import re +from threading import Lock +from typing import Any, List, Optional + from rag_system.utils.ollama_client import OllamaClient +# Serialises the first (heavy) load of a local verifier model, the same way the +# reranker and Provence loads are serialised in retrieval_pipeline.py. +_local_verifier_lock: Lock = Lock() + + class VerificationResult: - def __init__(self, is_grounded: bool, reasoning: str, verdict: str, confidence_score: int): + # The prompt still elicits `verdict`/`reasoning` strings, but nothing ever + # read them off this object โ€” only these two consumed fields are stored. + def __init__(self, is_grounded: bool, confidence_score: int): self.is_grounded = is_grounded - self.reasoning = reasoning - self.verdict = verdict self.confidence_score = confidence_score + +def _coerce_grounded(value: Any) -> bool: + """JSON ``true`` (or the string "true", any case) โ†’ True; anything else False. + + A verdict string of ``"false"`` must not end up truthy. + """ + if isinstance(value, bool): + return value + if isinstance(value, str): + return value.strip().lower() == "true" + return False + + +def _coerce_confidence(value: Any) -> int: + """Coerce a verdict's confidence_score to int; 0 for anything malformed. + + Accepts int/float (but not bool โ€” ``True`` is not a score) and numeric + strings via float(); null, arrays, objects and garbage all fail open to 0, + which the caller treats as "verifier unusable โ€” return the answer + unannotated". + """ + if isinstance(value, bool): + return 0 + if isinstance(value, (int, float)): + number = value + elif isinstance(value, str): + try: + number = float(value) + except ValueError: + return 0 + else: + return 0 + try: + return int(number) + except (OverflowError, ValueError): # NaN / inf survive lenient JSON parsing + return 0 + + +class VerifierModelUnavailable(RuntimeError): + """Raised when `VERIFIER_MODEL` names something that cannot be loaded.""" + + +# Models checked for suitability on 2026-08-09 (roadmap 2.4). Reported to the +# user verbatim when a configured verifier fails to load, so the failure names +# what was actually verified instead of hand-waving. +VERIFIER_AVAILABILITY_NOTES = """\ +Checked on 2026-08-09 (HuggingFace Hub API): + * ThinknCheck (arXiv 2604.01652, UPenn) โ€” NO PUBLIC WEIGHTS. The paper is + real (1B, 78.1 BAcc on LLMAggreFact) but a Hub search for "thinkncheck" + returns zero models and the paper links no release. Cannot be wired. + * ibm-granite/granite-guardian-3.3-8b โ€” exists, Apache-2.0, but 8B / + ~16 GB. Far over the "small local verifier" budget this seam is for. + * ibm-granite/granite-guardian-hap-38m โ€” exists, 38M, Apache-2.0, but it + is a hate/abuse/profanity RoBERTa classifier. Wrong task: it does not score + answer-vs-evidence entailment at all. +Verified working (<2 GB, no trust_remote_code): + * lytang/MiniCheck-DeBERTa-v3-Large (MIT, 1.74 GB) <- smoke-tested default + * lytang/MiniCheck-RoBERTa-Large (MIT) + * MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli (MIT, 369 MB, generic NLI) +Needs trust_remote_code (opt in with VERIFIER_TRUST_REMOTE_CODE=1): + * vectara/hallucination_evaluation_model (HHEM-2.1-open, Apache-2.0, 438 MB) +""" + +_SENTENCE_SPLIT = re.compile(r"(?<=[.!?])\s+") + + +class LocalNLIVerifier: + """Answer-vs-evidence scoring with a local sequence-classification model. + + The seam roadmap item 2.4 asks for. Any HuggingFace model that scores a + (premise, hypothesis) pair works: MiniCheck's grounded-claim checkers, a + generic MNLI cross-encoder, or Vectara's HHEM. The answer is split into + sentences, each is scored against the retrieved evidence as the premise, and + the **minimum** is taken โ€” one unsupported sentence makes the answer + ungrounded, which is the semantics the binary judge already uses. + + Note on ``[Confidence: N%]``: this number is a model output, not a + calibrated probability of correctness. It is UX, and `Documentation/ + verifier.md` says so. Swapping the LLM prompt for an NLI model changes + where the number comes from; it does not make it calibrated. + """ + + def __init__(self, model_name: str, threshold: float = 0.5, + trust_remote_code: Optional[bool] = None): + self.model_name = model_name + self.threshold = threshold + if trust_remote_code is None: + trust_remote_code = os.getenv("VERIFIER_TRUST_REMOTE_CODE", "") == "1" + try: + import torch + from transformers import AutoModelForSequenceClassification, AutoTokenizer + except ImportError as e: # pragma: no cover - transformers is a hard dep + raise VerifierModelUnavailable( + f"transformers/torch are required to load VERIFIER_MODEL: {e}") + + self._torch = torch + try: + self.tokenizer = AutoTokenizer.from_pretrained( + model_name, trust_remote_code=trust_remote_code) + self.model = AutoModelForSequenceClassification.from_pretrained( + model_name, trust_remote_code=trust_remote_code) + except Exception as e: + raise VerifierModelUnavailable( + f"Could not load VERIFIER_MODEL='{model_name}': {e}\n\n" + f"{VERIFIER_AVAILABILITY_NOTES}" + "Set VERIFIER_MODEL to one of the verified names above, unset it to " + "use the default LLM-prompt verifier, or add " + "VERIFIER_TRUST_REMOTE_CODE=1 if the model ships custom code." + ) from e + + self.model.eval() + self.device = ("mps" if torch.backends.mps.is_available() + else "cuda" if torch.cuda.is_available() else "cpu") + self.model.to(self.device) + self._supported_index = self._resolve_supported_index() + print(f"โœ… Local verifier '{model_name}' loaded on {self.device} " + f"(supported label index {self._supported_index}).") + + def _resolve_supported_index(self) -> int: + """Which logit means "the evidence supports this".""" + id2label = getattr(self.model.config, "id2label", None) or {} + for idx, label in id2label.items(): + if str(label).lower() in {"entailment", "consistent", "supported", "1", "true"}: + return int(idx) + # Binary checkers (MiniCheck) label their classes "0"/"1": 1 = supported. + return int(self.model.config.num_labels) - 1 + + def score(self, evidence: str, answer: str) -> float: + sentences = [s.strip() for s in _SENTENCE_SPLIT.split(answer or "") if s.strip()] + if not sentences: + return 0.0 + torch = self._torch + scores: List[float] = [] + with torch.no_grad(): + for sentence in sentences: + inputs = self.tokenizer(evidence, sentence, return_tensors="pt", + truncation=True, max_length=self.tokenizer.model_max_length + if self.tokenizer.model_max_length < 100000 else 2048) + inputs = {k: v.to(self.device) for k, v in inputs.items()} + logits = self.model(**inputs).logits[0] + if logits.numel() == 1: + probability = torch.sigmoid(logits)[0] + else: + probability = torch.softmax(logits, dim=-1)[self._supported_index] + scores.append(float(probability)) + # Weakest link: one unsupported sentence makes the answer ungrounded. + return min(scores) + + def verify(self, query: str, context: str, answer: str) -> VerificationResult: + probability = self.score(context, answer) + grounded = probability >= self.threshold + return VerificationResult( + is_grounded=grounded, + confidence_score=int(round(probability * 100)), + ) + + class Verifier: """ - Verifies if a generated answer is grounded in the provided context using Ollama. + Verifies if a generated answer is grounded in the provided context. + + Two backends, same interface: + + * **default** โ€” an LLM prompt on the utility model (below). This is what + ships; nothing changes unless you opt in. + * **local NLI/verifier model** โ€” set ``VERIFIER_MODEL`` (or + ``verification.model`` in the pipeline config) to a HuggingFace model name. + Loaded lazily on first use through ``LocalNLIVerifier``, so naming a model + costs nothing until a query is actually verified. A model that cannot be + loaded raises with the list of names that were checked, rather than + silently degrading to the LLM prompt โ€” a verifier that quietly is not the + verifier you configured is worse than an error. + + Roadmap item 2.4. Availability findings are in ``VERIFIER_AVAILABILITY_NOTES``. """ - def __init__(self, llm_client: OllamaClient, llm_model: str): + + def __init__(self, llm_client: OllamaClient, llm_model: str, + model_name: Optional[str] = None, threshold: float = 0.5): self.llm_client = llm_client self.llm_model = llm_model - print(f"Initialized Verifier with Ollama model '{self.llm_model}'.") + self.local_model_name = model_name or os.getenv("VERIFIER_MODEL") or None + self.local_threshold = threshold + self._local: Optional[LocalNLIVerifier] = None + if self.local_model_name: + print(f"Initialized Verifier with local model '{self.local_model_name}' " + f"(loaded on first use); LLM fallback model '{self.llm_model}'.") + else: + print(f"Initialized Verifier with Ollama model '{self.llm_model}'.") + + def _get_local(self) -> Optional[LocalNLIVerifier]: + if not self.local_model_name: + return None + if self._local is None: + with _local_verifier_lock: + if self._local is None: + self._local = LocalNLIVerifier(self.local_model_name, + self.local_threshold) + return self._local # Synchronous verify() method removed โ€“ async version is used everywhere. # --- Async wrapper ------------------------------------------------ async def verify_async(self, query: str, context: str, answer: str) -> VerificationResult: """Async variant that calls the Ollama client asynchronously.""" + local = self._get_local() + if local is not None: + # transformers is blocking; keep the event loop responsive. + return await asyncio.to_thread(local.verify, query, context, answer) + prompt = f""" You are an automated fact-checker. Determine whether the ANSWER is fully supported by the CONTEXT and output a single line of JSON. @@ -73,7 +279,15 @@ async def verify_async(self, query: str, context: str, answer: str) -> Verificat </QUERY> <CONTEXT> """ - prompt += context[:4000] # Clamp to avoid huge prompts + # Clamp to avoid huge prompts, and neutralize the prompt's own + # delimiter tags inside the retrieved evidence: an indexed document + # containing "</CONTEXT>" followed by a forged "<OUTPUT>" blob could + # otherwise steer the verdict (prompt injection via the corpus). The + # prompt structure itself is eval-tuned and must not change. + safe_context = context[:4000] + for tag in ("QUERY", "CONTEXT", "ANSWER", "OUTPUT"): + safe_context = safe_context.replace(f"<{tag}>", f"< {tag}>").replace(f"</{tag}>", f"< /{tag}>") + prompt += safe_context prompt += """ </CONTEXT> <ANSWER> @@ -83,14 +297,17 @@ async def verify_async(self, query: str, context: str, answer: str) -> Verificat </ANSWER> <OUTPUT> """ - resp = await self.llm_client.generate_completion_async(self.llm_model, prompt, format="json") + resp = await self.llm_client.generate_completion_async( + self.llm_model, prompt, format="json", + options={"temperature": 0}, # deterministic verdicts (same pin the eval judge got in ebcc88b) + ) try: - data = json.loads(resp.get("response", "{}")) + data = json.loads(resp.get("response") or "{}") + # Coerce defensively: a type-mismatched verdict (string "85", null, + # the string "false") must fail open โ€” never raise out of here. return VerificationResult( - is_grounded=data.get("is_grounded", False), - reasoning=data.get("reasoning", "async parse error"), - verdict=data.get("verdict", "NOT_SUPPORTED"), - confidence_score=data.get('confidence_score', 0) + is_grounded=_coerce_grounded(data.get("is_grounded", False)), + confidence_score=_coerce_confidence(data.get("confidence_score", 0)) ) except (json.JSONDecodeError, AttributeError): - return VerificationResult(False, "Failed async parse", "NOT_SUPPORTED", 0) + return VerificationResult(False, 0) diff --git a/rag_system/api_server.py b/rag_system/api_server.py index de148361..c98b33df 100644 --- a/rag_system/api_server.py +++ b/rag_system/api_server.py @@ -1,8 +1,11 @@ +import copy import json import http.server import socketserver -from urllib.parse import urlparse, parse_qs +from contextlib import contextmanager +from urllib.parse import urlparse import os +import re import requests import sys import logging @@ -12,112 +15,269 @@ if backend_dir not in sys.path: sys.path.append(backend_dir) -from backend.database import ChatDatabase, generate_session_title -from rag_system.main import get_agent -from rag_system.factory import get_indexing_pipeline +from backend.database import ChatDatabase +from rag_system.factory import get_agent, get_indexing_pipeline +from rag_system.main import LLM_BACKEND, PIPELINE_CONFIGS, WATSONX_CONFIG +from rag_system.retrieval.filters import FilterError, compile_filters -# Initialize database connection once at module level -# Use auto-detection for environment-appropriate path +logger = logging.getLogger(__name__) + +# The RAG API reads/writes index metadata only. Chat message rows are owned +# exclusively by backend/server.py. db = ChatDatabase() # Get the desired agent mode from environment variables, defaulting to 'default' -# This allows us to easily switch between 'default', 'fast', 'react', etc. AGENT_MODE = os.getenv("RAG_CONFIG_MODE", "default") + +# --- Global Singletons --- +# The agent and indexing pipeline are initialized once when the server starts so +# that models are not reloaded on every request. +print("๐Ÿง  Initializing RAG Agent... (This may take a moment)") RAG_AGENT = get_agent(AGENT_MODE) INDEXING_PIPELINE = get_indexing_pipeline(AGENT_MODE) +print("โœ… RAG Agent initialized successfully.") + +DEFAULT_TEXT_TABLE = PIPELINE_CONFIGS.get(AGENT_MODE, PIPELINE_CONFIGS["default"])["storage"]["text_table_name"] +SUPPORTED_RETRIEVAL_MODES = ("hybrid", "vector_only", "fts_only") +OLLAMA_TIMEOUT_SECONDS = 5 + +_CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])") + +# Wire aliases that are not a plain camelCase/snake_case pair. +_KEY_ALIASES = { + "overview_model": "overview_model_name", + "latechunk": "enable_latechunk", + "docling_chunk": "enable_docling_chunk", + "decompose": "query_decompose", +} + + +def normalize_request_keys(data): + """Accept camelCase and snake_case spellings of every option. + + The frontend historically sent camelCase while the backend gateway sends + snake_case. Both land in the same canonical snake_case key here, once, at + parse time. Explicit snake_case values always win over a camelCase twin. + """ + if not isinstance(data, dict): + return data + + normalized = {} + camel_derived = {} + for key, value in data.items(): + if not isinstance(key, str): + normalized[key] = value + continue + canonical = _CAMEL_BOUNDARY.sub("_", key).lower() + canonical = _KEY_ALIASES.get(canonical, canonical) + if canonical == key: + normalized[key] = value + else: + camel_derived[canonical] = value + + for key, value in camel_derived.items(): + normalized.setdefault(key, value) + return normalized + + +def _read_json_body(handler): + """Read and normalize a JSON request body. Returns an empty dict on empty body.""" + content_length = int(handler.headers.get('Content-Length') or 0) + if content_length <= 0: + return {} + post_data = handler.rfile.read(content_length) + return normalize_request_keys(json.loads(post_data.decode('utf-8'))) + + +def _model_valid_for_backend(model_name: str) -> bool: + """A watsonx deployment cannot serve an Ollama model id (and vice versa).""" + if LLM_BACKEND.lower() == "watsonx": + return "/" in model_name + return "/" not in model_name + + +@contextmanager +def _generation_model_override(requested_model): + """Apply a per-request generation model without permanently mutating the singleton.""" + applied = False + previous = RAG_AGENT.ollama_config.get("generation_model") + if isinstance(requested_model, str) and requested_model: + if _model_valid_for_backend(requested_model): + RAG_AGENT.ollama_config["generation_model"] = requested_model + applied = True + else: + logger.warning( + "Ignoring requested model '%s': not valid for the '%s' backend (using '%s').", + requested_model, LLM_BACKEND, previous, + ) + try: + yield + finally: + if applied: + RAG_AGENT.ollama_config["generation_model"] = previous -# --- Global Singleton for the RAG Agent --- -# The agent is initialized once when the server starts. -# This avoids reloading all the models on every request. -print("๐Ÿง  Initializing RAG Agent with MAXIMUM ACCURACY... (This may take a moment)") -if RAG_AGENT is None: - print("โŒ Critical error: RAG Agent could not be initialized. Exiting.") - exit(1) -print("โœ… RAG Agent initialized successfully with MAXIMUM ACCURACY.") -# --- - -# Add helper near top after db & agent init -# -------------- Helper ---------------- def _apply_index_embedding_model(idx_ids): - """Ensure retrieval pipeline uses the embedding model stored with the first index.""" - debug_info = f"๐Ÿ”ง _apply_index_embedding_model called with idx_ids: {idx_ids}\n" - + """Ensure the retrieval pipeline uses the embedding model of the active index. + + The active index is the LAST linked one (``idx_ids[-1]``), matching the + gateway, which builds the retrieval table name from the same element. + ``get_indexes_for_session`` returns links oldest-first, so the last entry + is the most recently linked index. + """ if not idx_ids: - debug_info += "โš ๏ธ No index IDs provided\n" - with open("logs/embedding_debug.log", "a") as f: - f.write(debug_info) + logger.debug("No index IDs provided; keeping the configured embedding model.") return try: - idx = db.get_index(idx_ids[0]) - debug_info += f"๐Ÿ”ง Retrieved index: {idx.get('id')} with metadata: {idx.get('metadata', {})}\n" + idx = db.get_index(idx_ids[-1]) model = (idx.get("metadata") or {}).get("embedding_model") - debug_info += f"๐Ÿ”ง Embedding model from metadata: {model}\n" - if model: - rp = RAG_AGENT.retrieval_pipeline - current_model = rp.config.get("embedding_model_name") - debug_info += f"๐Ÿ”ง Current embedding model: {current_model}\n" - rp.update_embedding_model(model) - debug_info += f"๐Ÿ”ง Updated embedding model to: {model}\n" - else: - debug_info += "โš ๏ธ No embedding model found in metadata\n" + if not model: + logger.debug("Index %s has no embedding_model metadata.", idx_ids[-1]) + return + rp = RAG_AGENT.retrieval_pipeline + logger.info( + "Applying index embedding model '%s' (was '%s').", + model, rp.config.get("embedding_model_name"), + ) + rp.update_embedding_model(model) except Exception as e: - debug_info += f"โš ๏ธ Could not apply index embedding model: {e}\n" - - # Write debug info to file - with open("logs/embedding_debug.log", "a") as f: - f.write(debug_info) + logger.warning("Could not apply index embedding model: %s", e) + def _get_table_name_for_session(session_id): """Get the correct vector table name for a session by looking up its linked indexes.""" - logger = logging.getLogger(__name__) - if not session_id: - logger.info("โŒ No session_id provided") return None - + try: - # Get indexes linked to this session idx_ids = db.get_indexes_for_session(session_id) - logger.info(f"๐Ÿ” Session {session_id[:8]}... has {len(idx_ids)} indexes: {idx_ids}") - if not idx_ids: - logger.warning(f"โš ๏ธ No indexes found for session {session_id}") - # Use the default table name from config instead of session-specific name - from rag_system.main import PIPELINE_CONFIGS - default_table = PIPELINE_CONFIGS["default"]["storage"]["text_table_name"] - logger.info(f"๐Ÿ“Š Using default table '{default_table}' for session {session_id[:8]}...") - return default_table - - # Use the first index's vector table name - idx = db.get_index(idx_ids[0]) + logger.info("No indexes for session %s; using default table '%s'.", session_id, DEFAULT_TEXT_TABLE) + return DEFAULT_TEXT_TABLE + + # The active index is the LAST linked one, same as the gateway's + # table-name resolution and _apply_index_embedding_model above. + idx = db.get_index(idx_ids[-1]) if idx and idx.get('vector_table_name'): table_name = idx['vector_table_name'] - logger.info(f"๐Ÿ“Š Using table '{table_name}' for session {session_id[:8]}...") - print(f"๐Ÿ“Š RAG API: Using table '{table_name}' for session {session_id[:8]}...") + logger.info("Using table '%s' for session %s.", table_name, session_id) return table_name - else: - logger.warning(f"โš ๏ธ Index found but no vector table name for session {session_id}") - # Use the default table name from config instead of session-specific name - from rag_system.main import PIPELINE_CONFIGS - default_table = PIPELINE_CONFIGS["default"]["storage"]["text_table_name"] - logger.info(f"๐Ÿ“Š Using default table '{default_table}' for session {session_id[:8]}...") - return default_table - + + logger.warning("Index found but no vector table name for session %s.", session_id) + return DEFAULT_TEXT_TABLE except Exception as e: - logger.error(f"โŒ Error getting table name for session {session_id}: {e}") - # Use the default table name from config instead of session-specific name - from rag_system.main import PIPELINE_CONFIGS - default_table = PIPELINE_CONFIGS["default"]["storage"]["text_table_name"] - logger.info(f"๐Ÿ“Š Using default table '{default_table}' for session {session_id[:8]}...") - return default_table + logger.error("Error getting table name for session %s: %s", session_id, e) + return DEFAULT_TEXT_TABLE + + +def _resolve_retrieval_mode(data): + """`retrieval_mode` is the wire name for the pipeline's `search_type`; both are accepted.""" + value = data.get('retrieval_mode') or data.get('search_type') + if value is None: + return None, None + if value not in SUPPORTED_RETRIEVAL_MODES: + return None, f"Unsupported retrieval mode '{value}'. Supported: {', '.join(SUPPORTED_RETRIEVAL_MODES)}." + return value, None + + +def _resolve_filters(data): + """Compile the optional ``filters`` object (roadmap item 4.4). + + Returns ``(compiled_or_None, error_or_None)``. A filter that cannot be + compiled is a 400 and never a silently-unfiltered search: the caller asked + to be restricted to part of the corpus, and answering from all of it would + be the wrong answer delivered confidently. + """ + raw = data.get('filters') + if raw is None: + return None, None + try: + return compile_filters(raw), None + except FilterError as e: + return None, f"Invalid filters: {e}" + + +def _parse_chat_request(data): + """Extract the canonical chat options from an already-normalized body.""" + retrieval_mode, error = _resolve_retrieval_mode(data) + if error: + return None, error + + filters, error = _resolve_filters(data) + if error: + return None, error + + return { + "filters": filters, + "query": data.get('query'), + "session_id": data.get('session_id'), + "table_name": data.get('table_name'), + "model": data.get('model'), + "compose_sub_answers": data.get('compose_sub_answers'), + "query_decompose": data.get('query_decompose'), + "ai_rerank": data.get('ai_rerank'), + "context_expand": data.get('context_expand'), + "verify": data.get('verify'), + "retrieval_k": data.get('retrieval_k'), + "context_window_size": data.get('context_window_size'), + "reranker_top_k": data.get('reranker_top_k'), + "retrieval_mode": retrieval_mode, + "force_rag": bool(data.get('force_rag', False)), + "provence_prune": data.get('provence_prune'), + "provence_threshold": data.get('provence_threshold'), + }, None + + +def _build_index_config_override(base_config, *, table_name, options): + """Build the per-request indexing config from the pipeline profile plus request options.""" + config_override = copy.deepcopy(base_config) + retrieval_cfg = config_override.setdefault("retrieval", {}) + + if table_name: + config_override.setdefault("storage", {})["text_table_name"] = table_name + retrieval_cfg.setdefault("dense", {})["lancedb_table_name"] = table_name + + retrieval_cfg.setdefault("latechunk", {})["enabled"] = options["enable_latechunk"] + + # `retrieval_mode` is the wire name for the pipeline's `search_type`; it is + # recorded on the index config so the built index carries the mode it was + # requested with. + if options["retrieval_mode"] is not None: + retrieval_cfg["search_type"] = options["retrieval_mode"] + + config_override["chunker_mode"] = "docling" if options["enable_docling_chunk"] else "legacy" + + enricher_cfg = config_override.setdefault("contextual_enricher", {}) + enricher_cfg["enabled"] = options["enable_enrich"] + enricher_cfg["window_size"] = options["window_size"] + + indexing_cfg = config_override.setdefault("indexing", {}) + indexing_cfg["embedding_batch_size"] = options["batch_size_embed"] + indexing_cfg["enrichment_batch_size"] = options["batch_size_enrich"] + + config_override.setdefault("chunking", {})["chunk_size"] = options["chunk_size"] + + if options["embedding_model"]: + config_override["embedding_model_name"] = options["embedding_model"] + if options["enrich_model"]: + config_override["enrich_model"] = options["enrich_model"] + if options["overview_model_name"]: + config_override["overview_model_name"] = options["overview_model_name"] + if options["session_id"]: + config_override["overview_path"] = f"index_store/overviews/{options['session_id']}.jsonl" + + return config_override + class AdvancedRagApiHandler(http.server.BaseHTTPRequestHandler): + def log_message(self, format, *args): + logger.info("%s - %s", self.address_string(), format % args) + def do_OPTIONS(self): """Handle CORS preflight requests for frontend integration.""" self.send_response(200) self.send_header('Access-Control-Allow-Origin', '*') - self.send_header('Access-Control-Allow-Methods', 'POST, OPTIONS') + self.send_header('Access-Control-Allow-Methods', 'GET, POST, OPTIONS') self.send_header('Access-Control-Allow-Headers', 'Content-Type') self.end_headers() @@ -137,7 +297,9 @@ def do_POST(self): def do_GET(self): parsed_path = urlparse(self.path) - if parsed_path.path == '/models': + if parsed_path.path == '/health': + self.send_json_response({"status": "ok"}) + elif parsed_path.path == '/models': self.handle_models() else: self.send_json_response({"error": "Not Found"}, status_code=404) @@ -145,234 +307,47 @@ def do_GET(self): def handle_chat(self): """Handles a chat query by calling the agentic RAG pipeline.""" try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - - query = data.get('query') - session_id = data.get('session_id') - compose_flag = data.get('compose_sub_answers') - decomp_flag = data.get('query_decompose') - ai_rerank_flag = data.get('ai_rerank') - ctx_expand_flag = data.get('context_expand') - verify_flag = data.get('verify') - - # โœจ NEW RETRIEVAL PARAMETERS - retrieval_k = data.get('retrieval_k', 20) - context_window_size = data.get('context_window_size', 1) - reranker_top_k = data.get('reranker_top_k', 10) - search_type = data.get('search_type', 'hybrid') - dense_weight = data.get('dense_weight', 0.7) - - # ๐Ÿšฉ NEW: Force RAG override from frontend - force_rag = bool(data.get('force_rag', False)) - - # ๐ŸŒฟ Provence sentence pruning - provence_prune = data.get('provence_prune') - provence_threshold = data.get('provence_threshold') - - # User-selected generation model - requested_model = data.get('model') - if isinstance(requested_model,str) and requested_model: - RAG_AGENT.ollama_config['generation_model']=requested_model - + data = _read_json_body(self) + params, error = _parse_chat_request(data) + if error: + self.send_json_response({"error": error}, status_code=400) + return + + query = params["query"] if not query: self.send_json_response({"error": "Query is required"}, status_code=400) return - # ๐Ÿ”„ UPDATE SESSION TITLE: If this is the first message in the session, update the title - if session_id: - try: - # Check if this is the first message by calling the backend server - backend_url = f"http://localhost:8000/sessions/{session_id}" - session_resp = requests.get(backend_url) - if session_resp.status_code == 200: - session_data = session_resp.json() - session = session_data.get('session', {}) - # If message_count is 0, this is the first message - if session.get('message_count', 0) == 0: - # Generate a title from the first message - title = generate_session_title(query) - # Update the session title via backend API - # We'll need to add this endpoint to the backend, for now let's make a direct database call - # This is a temporary solution until we add a proper API endpoint - db.update_session_title(session_id, title) - print(f"๐Ÿ“ Updated session title to: {title}") - - # ๐Ÿ’พ STORE USER MESSAGE: Add the user message to the database - user_message_id = db.add_message(session_id, query, "user") - print(f"๐Ÿ’พ Stored user message: {user_message_id}") - else: - # Not the first message, but still store the user message - user_message_id = db.add_message(session_id, query, "user") - print(f"๐Ÿ’พ Stored user message: {user_message_id}") - except Exception as e: - print(f"โš ๏ธ Failed to update session title or store user message: {e}") - # Continue with the request even if title update fails - - # Allow explicit table_name override - table_name = data.get('table_name') - if not table_name and session_id: - table_name = _get_table_name_for_session(session_id) - - # Decide execution path - print(f"๐Ÿ”ง Force RAG flag: {force_rag}") - if force_rag: - # --- Apply runtime overrides manually because we skip Agent.run() - rp_cfg = RAG_AGENT.retrieval_pipeline.config - if retrieval_k is not None: - rp_cfg["retrieval_k"] = retrieval_k - if reranker_top_k is not None: - rp_cfg.setdefault("reranker", {})["top_k"] = reranker_top_k - if search_type is not None: - rp_cfg.setdefault("retrieval", {})["search_type"] = search_type - if dense_weight is not None: - rp_cfg.setdefault("retrieval", {}).setdefault("dense", {})["weight"] = dense_weight - - # Provence overrides - if provence_prune is not None: - rp_cfg.setdefault("provence", {})["enabled"] = bool(provence_prune) - if provence_threshold is not None: - rp_cfg.setdefault("provence", {})["threshold"] = float(provence_threshold) - - # ๐Ÿ”„ Apply embedding model for this session (same as in agent path) - if session_id: - idx_ids = db.get_indexes_for_session(session_id) - _apply_index_embedding_model(idx_ids) - - # Directly invoke retrieval pipeline to bypass triage - result = RAG_AGENT.retrieval_pipeline.run( - query, - table_name=table_name, - window_size_override=context_window_size, - ) - else: - # Use full agent with smart routing - # Apply Provence overrides even in agent path - rp_cfg = RAG_AGENT.retrieval_pipeline.config - if provence_prune is not None: - rp_cfg.setdefault("provence", {})["enabled"] = bool(provence_prune) - if provence_threshold is not None: - rp_cfg.setdefault("provence", {})["threshold"] = float(provence_threshold) - - # ๐Ÿ”„ Refresh document overviews for this session - if session_id: - idx_ids = db.get_indexes_for_session(session_id) - _apply_index_embedding_model(idx_ids) - RAG_AGENT.load_overviews_for_indexes(idx_ids) - - # ๐Ÿ”ง Set index-specific overview path - if session_id: - rp_cfg["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # ๐Ÿ”ง Configure late chunking - rp_cfg.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - result = RAG_AGENT.run( - query, - table_name=table_name, - session_id=session_id, - compose_sub_answers=compose_flag, - query_decompose=decomp_flag, - ai_rerank=ai_rerank_flag, - context_expand=ctx_expand_flag, - verify=verify_flag, - retrieval_k=retrieval_k, - context_window_size=context_window_size, - reranker_top_k=reranker_top_k, - search_type=search_type, - dense_weight=dense_weight, - ) - - # The result is a dict, so we need to dump it to a JSON string + session_id = params["session_id"] + table_name = params["table_name"] or _get_table_name_for_session(session_id) + + with _generation_model_override(params["model"]): + result = self._run_query(params, query, table_name, session_id, emit=None) + self.send_json_response(result) - - # ๐Ÿ’พ STORE AI RESPONSE: Add the AI response to the database - if session_id and result and result.get("answer"): - try: - ai_message_id = db.add_message(session_id, result["answer"], "assistant") - print(f"๐Ÿ’พ Stored AI response: {ai_message_id}") - except Exception as e: - print(f"โš ๏ธ Failed to store AI response: {e}") - # Continue even if storage fails except json.JSONDecodeError: self.send_json_response({"error": "Invalid JSON"}, status_code=400) except Exception as e: + logger.exception("Chat request failed") self.send_json_response({"error": f"Server error: {str(e)}"}, status_code=500) def handle_chat_stream(self): """Stream internal phases and final answer using SSE (text/event-stream).""" try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - - query = data.get('query') - session_id = data.get('session_id') - compose_flag = data.get('compose_sub_answers') - decomp_flag = data.get('query_decompose') - ai_rerank_flag = data.get('ai_rerank') - ctx_expand_flag = data.get('context_expand') - verify_flag = data.get('verify') - - # โœจ NEW RETRIEVAL PARAMETERS - retrieval_k = data.get('retrieval_k', 20) - context_window_size = data.get('context_window_size', 1) - reranker_top_k = data.get('reranker_top_k', 10) - search_type = data.get('search_type', 'hybrid') - dense_weight = data.get('dense_weight', 0.7) - - # ๐Ÿšฉ NEW: Force RAG override from frontend - force_rag = bool(data.get('force_rag', False)) - - # ๐ŸŒฟ Provence sentence pruning - provence_prune = data.get('provence_prune') - provence_threshold = data.get('provence_threshold') - - # User-selected generation model - requested_model = data.get('model') - if isinstance(requested_model,str) and requested_model: - RAG_AGENT.ollama_config['generation_model']=requested_model + data = _read_json_body(self) + params, error = _parse_chat_request(data) + if error: + self.send_json_response({"error": error}, status_code=400) + return + query = params["query"] if not query: self.send_json_response({"error": "Query is required"}, status_code=400) return - # ๐Ÿ”„ UPDATE SESSION TITLE: If this is the first message in the session, update the title - if session_id: - try: - # Check if this is the first message by calling the backend server - backend_url = f"http://localhost:8000/sessions/{session_id}" - session_resp = requests.get(backend_url) - if session_resp.status_code == 200: - session_data = session_resp.json() - session = session_data.get('session', {}) - # If message_count is 0, this is the first message - if session.get('message_count', 0) == 0: - # Generate a title from the first message - title = generate_session_title(query) - # Update the session title via backend API - # We'll need to add this endpoint to the backend, for now let's make a direct database call - # This is a temporary solution until we add a proper API endpoint - db.update_session_title(session_id, title) - print(f"๐Ÿ“ Updated session title to: {title}") - - # ๐Ÿ’พ STORE USER MESSAGE: Add the user message to the database - user_message_id = db.add_message(session_id, query, "user") - print(f"๐Ÿ’พ Stored user message: {user_message_id}") - else: - # Not the first message, but still store the user message - user_message_id = db.add_message(session_id, query, "user") - print(f"๐Ÿ’พ Stored user message: {user_message_id}") - except Exception as e: - print(f"โš ๏ธ Failed to update session title or store user message: {e}") - # Continue with the request even if title update fails - - # Allow explicit table_name override - table_name = data.get('table_name') - if not table_name and session_id: - table_name = _get_table_name_for_session(session_id) + session_id = params["session_id"] + table_name = params["table_name"] or _get_table_name_for_session(session_id) # Prepare response headers for SSE self.send_response(200) @@ -386,347 +361,198 @@ def handle_chat_stream(self): def emit(event_type: str, payload): """Send a single SSE event.""" - try: - data_str = json.dumps({"type": event_type, "data": payload}) - self.wfile.write(f"data: {data_str}\n\n".encode('utf-8')) - self.wfile.flush() - except BrokenPipeError: - # Client disconnected - raise + data_str = json.dumps({"type": event_type, "data": payload}) + self.wfile.write(f"data: {data_str}\n\n".encode('utf-8')) + self.wfile.flush() - # Run the agent synchronously, emitting checkpoints try: - if force_rag: - # Apply overrides same as above since we bypass Agent.run - rp_cfg = RAG_AGENT.retrieval_pipeline.config - if retrieval_k is not None: - rp_cfg["retrieval_k"] = retrieval_k - if reranker_top_k is not None: - rp_cfg.setdefault("reranker", {})["top_k"] = reranker_top_k - if search_type is not None: - rp_cfg.setdefault("retrieval", {})["search_type"] = search_type - if dense_weight is not None: - rp_cfg.setdefault("retrieval", {}).setdefault("dense", {})["weight"] = dense_weight - - # Provence overrides - if provence_prune is not None: - rp_cfg.setdefault("provence", {})["enabled"] = bool(provence_prune) - if provence_threshold is not None: - rp_cfg.setdefault("provence", {})["threshold"] = float(provence_threshold) - - # ๐Ÿ”„ Apply embedding model for this session (same as in agent path) - if session_id: - idx_ids = db.get_indexes_for_session(session_id) - _apply_index_embedding_model(idx_ids) - - # ๐Ÿ”ง Set index-specific overview path so each index writes separate file - if session_id: - rp_cfg["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # ๐Ÿ”ง Configure late chunking - rp_cfg.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - # Straight retrieval pipeline with streaming events - final_result = RAG_AGENT.retrieval_pipeline.run( - query, - table_name=table_name, - window_size_override=context_window_size, - event_callback=emit, - ) - else: - # Provence overrides - rp_cfg = RAG_AGENT.retrieval_pipeline.config - if provence_prune is not None: - rp_cfg.setdefault("provence", {})["enabled"] = bool(provence_prune) - if provence_threshold is not None: - rp_cfg.setdefault("provence", {})["threshold"] = float(provence_threshold) - - # ๐Ÿ”„ Refresh overviews for this session - if session_id: - idx_ids = db.get_indexes_for_session(session_id) - _apply_index_embedding_model(idx_ids) - RAG_AGENT.load_overviews_for_indexes(idx_ids) - - # ๐Ÿ”ง Set index-specific overview path - if session_id: - rp_cfg["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # ๐Ÿ”ง Configure late chunking - rp_cfg.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - final_result = RAG_AGENT.run( - query, - table_name=table_name, - session_id=session_id, - compose_sub_answers=compose_flag, - query_decompose=decomp_flag, - ai_rerank=ai_rerank_flag, - context_expand=ctx_expand_flag, - verify=verify_flag, - # โœจ NEW RETRIEVAL PARAMETERS - retrieval_k=retrieval_k, - context_window_size=context_window_size, - reranker_top_k=reranker_top_k, - search_type=search_type, - dense_weight=dense_weight, - event_callback=emit, - ) - - # Ensure the final answer is sent (in case callback missed it) + with _generation_model_override(params["model"]): + final_result = self._run_query(params, query, table_name, session_id, emit=emit) emit("complete", final_result) - - # ๐Ÿ’พ STORE AI RESPONSE: Add the AI response to the database - if session_id and final_result and final_result.get("answer"): - try: - ai_message_id = db.add_message(session_id, final_result["answer"], "assistant") - print(f"๐Ÿ’พ Stored AI response: {ai_message_id}") - except Exception as e: - print(f"โš ๏ธ Failed to store AI response: {e}") - # Continue even if storage fails except BrokenPipeError: - print("๐Ÿ”Œ Client disconnected from SSE stream.") + logger.info("Client disconnected from SSE stream.") except Exception as e: - # Send error event then close - error_payload = {"error": str(e)} + logger.exception("Stream error") try: - emit("error", error_payload) - finally: - print(f"โŒ Stream error: {e}") + emit("error", {"error": str(e)}) + except BrokenPipeError: + pass except json.JSONDecodeError: self.send_json_response({"error": "Invalid JSON"}, status_code=400) except Exception as e: + logger.exception("Chat stream request failed") self.send_json_response({"error": f"Server error: {str(e)}"}, status_code=500) + def _run_query(self, params, query, table_name, session_id, emit): + """Shared execution path for /chat and /chat/stream. + + Everything, including force_rag, goes through Agent.run so that the + verify / ai_rerank / decompose toggles are honored on every path. + """ + rp_cfg = RAG_AGENT.retrieval_pipeline.config + + if session_id: + idx_ids = db.get_indexes_for_session(session_id) + _apply_index_embedding_model(idx_ids) + rp_cfg["overview_path"] = f"index_store/overviews/{session_id}.jsonl" + RAG_AGENT.load_overviews_for_indexes(idx_ids) + + if params["provence_prune"] is not None: + rp_cfg.setdefault("provence", {})["enabled"] = bool(params["provence_prune"]) + if params["provence_threshold"] is not None: + rp_cfg.setdefault("provence", {})["threshold"] = float(params["provence_threshold"]) + + run_kwargs = { + "table_name": table_name, + "session_id": session_id, + "compose_sub_answers": params["compose_sub_answers"], + "query_decompose": params["query_decompose"], + "ai_rerank": params["ai_rerank"], + "context_expand": params["context_expand"], + "verify": params["verify"], + "retrieval_k": params["retrieval_k"], + "context_window_size": params["context_window_size"], + "reranker_top_k": params["reranker_top_k"], + "retrieval_mode": params["retrieval_mode"], + "force_rag": params["force_rag"], + } + if params["filters"] is not None: + # Only added when present, so an unfiltered request reaches + # Agent.run with exactly the arguments it always did. + run_kwargs["filters"] = params["filters"] + if emit is not None: + run_kwargs["event_callback"] = emit + + return RAG_AGENT.run(query, **run_kwargs) + def handle_index(self): """Triggers the document indexing pipeline for specific files.""" try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - + data = _read_json_body(self) + file_paths = data.get('file_paths') - session_id = data.get('session_id') - compose_flag = data.get('compose_sub_answers') - decomp_flag = data.get('query_decompose') - ai_rerank_flag = data.get('ai_rerank') - ctx_expand_flag = data.get('context_expand') - enable_latechunk = bool(data.get("enable_latechunk", False)) - enable_docling_chunk = bool(data.get("enable_docling_chunk", False)) - - # ๐Ÿ†• NEW CONFIGURATION OPTIONS: - chunk_size = int(data.get("chunk_size", 512)) - chunk_overlap = int(data.get("chunk_overlap", 64)) - retrieval_mode = data.get("retrieval_mode", "hybrid") - window_size = int(data.get("window_size", 2)) - enable_enrich = bool(data.get("enable_enrich", True)) - embedding_model = data.get('embeddingModel') - enrich_model = data.get('enrichModel') - overview_model = data.get('overviewModel') or data.get('overview_model_name') - batch_size_embed = int(data.get("batch_size_embed", 50)) - batch_size_enrich = int(data.get("batch_size_enrich", 25)) - if not file_paths or not isinstance(file_paths, list): - self.send_json_response({ - "error": "A 'file_paths' list is required." - }, status_code=400) + self.send_json_response({"error": "A 'file_paths' list is required."}, status_code=400) + return + + retrieval_mode, error = _resolve_retrieval_mode(data) + if error: + self.send_json_response({"error": error}, status_code=400) return - # Allow explicit table_name override - table_name = data.get('table_name') - if not table_name and session_id: - table_name = _get_table_name_for_session(session_id) - - # The INDEXING_PIPELINE is already initialized. We just need to use it. - # If a session-specific table is needed, we can override the config for this run. - if table_name: - import copy - config_override = copy.deepcopy(INDEXING_PIPELINE.config) - config_override["storage"]["text_table_name"] = table_name - config_override.setdefault("retrievers", {}).setdefault("dense", {})["lancedb_table_name"] = table_name - - # ๐Ÿ”ง Configure late chunking - if enable_latechunk: - config_override["retrievers"].setdefault("latechunk", {})["enabled"] = True + session_id = data.get('session_id') + options = { + "session_id": session_id, + "enable_latechunk": bool(data.get("enable_latechunk", False)), + "enable_docling_chunk": bool(data.get("enable_docling_chunk", True)), + "chunk_size": int(data.get("chunk_size", 512)), + "retrieval_mode": retrieval_mode, + "window_size": int(data.get("window_size", 2)), + "enable_enrich": bool(data.get("enable_enrich", True)), + "embedding_model": data.get('embedding_model'), + "enrich_model": data.get('enrich_model'), + "overview_model_name": data.get('overview_model_name'), + "batch_size_embed": int(data.get("batch_size_embed", 50)), + "batch_size_enrich": int(data.get("batch_size_enrich", 25)), + } + + table_name = data.get('table_name') or _get_table_name_for_session(session_id) + + config_override = _build_index_config_override( + INDEXING_PIPELINE.config, table_name=table_name, options=options + ) + + logger.info( + "Indexing %d file(s) | table=%s | enrich=%s (window %s) | latechunk=%s | " + "chunk_size=%s | embedding=%s | enrichment=%s", + len(file_paths), table_name or DEFAULT_TEXT_TABLE, options["enable_enrich"], + options["window_size"], options["enable_latechunk"], options["chunk_size"], + options["embedding_model"] or config_override.get("embedding_model_name"), + options["enrich_model"] or "default", + ) + + temp_pipeline = INDEXING_PIPELINE.__class__( + config_override, + INDEXING_PIPELINE.llm_client, + INDEXING_PIPELINE.ollama_config, + ) + temp_pipeline.run(file_paths) + + if options["embedding_model"] and session_id: + # session_id is a chat-session UUID for session-scoped builds, + # which has no row in the indexes table; only merge metadata + # when it actually names an index. + if db.get_index(session_id): + try: + db.update_index_metadata(session_id, {"embedding_model": options["embedding_model"]}) + except Exception as e: + logger.warning("Could not update embedding_model metadata: %s", e) else: - # ensure disabled if not requested - config_override["retrievers"].setdefault("latechunk", {})["enabled"] = False - - # ๐Ÿ”ง Configure docling chunking - if enable_docling_chunk: - config_override["chunker_mode"] = "docling" - - # ๐Ÿ”ง Configure contextual enrichment (THIS WAS MISSING!) - config_override.setdefault("contextual_enricher", {}) - config_override["contextual_enricher"]["enabled"] = enable_enrich - config_override["contextual_enricher"]["window_size"] = window_size - - # ๐Ÿ”ง Configure indexing batch sizes - config_override.setdefault("indexing", {}) - config_override["indexing"]["embedding_batch_size"] = batch_size_embed - config_override["indexing"]["enrichment_batch_size"] = batch_size_enrich - - # ๐Ÿ”ง Configure chunking parameters - config_override.setdefault("chunking", {}) - config_override["chunking"]["chunk_size"] = chunk_size - config_override["chunking"]["chunk_overlap"] = chunk_overlap - - # ๐Ÿ”ง Configure embedding model if specified - if embedding_model: - config_override["embedding_model_name"] = embedding_model - - # ๐Ÿ”ง Configure enrichment model if specified - if enrich_model: - config_override["enrich_model"] = enrich_model - - # ๐Ÿ”ง Overview model (can differ from enrichment) - if overview_model: - config_override["overview_model_name"] = overview_model - - print(f"๐Ÿ”ง INDEXING CONFIG: Contextual Enrichment: {enable_enrich}, Window Size: {window_size}") - print(f"๐Ÿ”ง CHUNKING CONFIG: Size: {chunk_size}, Overlap: {chunk_overlap}") - print(f"๐Ÿ”ง MODEL CONFIG: Embedding: {embedding_model or 'default'}, Enrichment: {enrich_model or 'default'}") - - # ๐Ÿ”ง Set index-specific overview path so each index writes separate file - if session_id: - config_override["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # ๐Ÿ”ง Configure late chunking - config_override.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - # Create a temporary pipeline instance with the overridden config - temp_pipeline = INDEXING_PIPELINE.__class__( - config_override, - INDEXING_PIPELINE.llm_client, - INDEXING_PIPELINE.ollama_config - ) - temp_pipeline.run(file_paths) - else: - # Use the default pipeline with overrides - import copy - config_override = copy.deepcopy(INDEXING_PIPELINE.config) - - # ๐Ÿ”ง Configure late chunking - if enable_latechunk: - config_override.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - # ๐Ÿ”ง Configure docling chunking - if enable_docling_chunk: - config_override["chunker_mode"] = "docling" - - # ๐Ÿ”ง Configure contextual enrichment (THIS WAS MISSING!) - config_override.setdefault("contextual_enricher", {}) - config_override["contextual_enricher"]["enabled"] = enable_enrich - config_override["contextual_enricher"]["window_size"] = window_size - - # ๐Ÿ”ง Configure indexing batch sizes - config_override.setdefault("indexing", {}) - config_override["indexing"]["embedding_batch_size"] = batch_size_embed - config_override["indexing"]["enrichment_batch_size"] = batch_size_enrich - - # ๐Ÿ”ง Configure chunking parameters - config_override.setdefault("chunking", {}) - config_override["chunking"]["chunk_size"] = chunk_size - config_override["chunking"]["chunk_overlap"] = chunk_overlap - - # ๐Ÿ”ง Configure embedding model if specified - if embedding_model: - config_override["embedding_model_name"] = embedding_model - - # ๐Ÿ”ง Configure enrichment model if specified - if enrich_model: - config_override["enrich_model"] = enrich_model - - # ๐Ÿ”ง Overview model (can differ from enrichment) - if overview_model: - config_override["overview_model_name"] = overview_model - - print(f"๐Ÿ”ง INDEXING CONFIG: Contextual Enrichment: {enable_enrich}, Window Size: {window_size}") - print(f"๐Ÿ”ง CHUNKING CONFIG: Size: {chunk_size}, Overlap: {chunk_overlap}") - print(f"๐Ÿ”ง MODEL CONFIG: Embedding: {embedding_model or 'default'}, Enrichment: {enrich_model or 'default'}") - - # ๐Ÿ”ง Set index-specific overview path so each index writes separate file - if session_id: - config_override["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # ๐Ÿ”ง Configure late chunking - config_override.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - # Create temporary pipeline with overridden config - temp_pipeline = INDEXING_PIPELINE.__class__( - config_override, - INDEXING_PIPELINE.llm_client, - INDEXING_PIPELINE.ollama_config - ) - temp_pipeline.run(file_paths) + logger.debug("Skipping embedding_model metadata update: %s is not an index.", session_id) self.send_json_response({ "message": f"Indexing process for {len(file_paths)} file(s) completed successfully.", - "table_name": table_name or "default_text_table", - "latechunk": enable_latechunk, - "docling_chunk": enable_docling_chunk, + "table_name": table_name or DEFAULT_TEXT_TABLE, + "latechunk": options["enable_latechunk"], + "docling_chunk": options["enable_docling_chunk"], "indexing_config": { - "chunk_size": chunk_size, - "chunk_overlap": chunk_overlap, - "retrieval_mode": retrieval_mode, - "window_size": window_size, - "enable_enrich": enable_enrich, - "embedding_model": embedding_model, - "enrich_model": enrich_model, - "batch_size_embed": batch_size_embed, - "batch_size_enrich": batch_size_enrich + "chunk_size": options["chunk_size"], + "retrieval_mode": config_override.get("retrieval", {}).get("search_type"), + "window_size": options["window_size"], + "enable_enrich": options["enable_enrich"], + "embedding_model": config_override.get("embedding_model_name"), + "enrich_model": options["enrich_model"], + "overview_model_name": options["overview_model_name"], + "batch_size_embed": options["batch_size_embed"], + "batch_size_enrich": options["batch_size_enrich"], } }) - if embedding_model: - try: - db.update_index_metadata(session_id, {"embedding_model": embedding_model}) - except Exception as e: - print(f"โš ๏ธ Could not update embedding_model metadata: {e}") - except json.JSONDecodeError: self.send_json_response({"error": "Invalid JSON"}, status_code=400) except Exception as e: + logger.exception("Indexing request failed") self.send_json_response({"error": f"Failed to start indexing: {str(e)}"}, status_code=500) def handle_models(self): - """Return a list of locally installed Ollama models and supported HuggingFace models, grouped by capability.""" + """Return the generation and embedding models available to the active backend.""" try: generation_models = [] - embedding_models = [] - - # Get Ollama models if available - try: - resp = requests.get(f"{RAG_AGENT.ollama_config['host']}/api/tags", timeout=5) - resp.raise_for_status() - data = resp.json() - - all_ollama_models = [m.get('name') for m in data.get('models', [])] - - # Very naive classification - ollama_embedding_models = [m for m in all_ollama_models if any(k in m for k in ['embed','bge','embedding','text'])] - ollama_generation_models = [m for m in all_ollama_models if m not in ollama_embedding_models] - - generation_models.extend(ollama_generation_models) - embedding_models.extend(ollama_embedding_models) - except Exception as e: - print(f"โš ๏ธ Could not get Ollama models: {e}") - - # Add supported HuggingFace embedding models - huggingface_embedding_models = [ + embedding_models = [RAG_AGENT.retrieval_pipeline.config.get("embedding_model_name")] + + if LLM_BACKEND.lower() == "watsonx": + generation_models.extend( + m for m in (WATSONX_CONFIG.get("generation_model"), WATSONX_CONFIG.get("enrichment_model")) if m + ) + else: + host = RAG_AGENT.ollama_config.get('host') + try: + resp = requests.get(f"{host}/api/tags", timeout=OLLAMA_TIMEOUT_SECONDS) + resp.raise_for_status() + all_ollama_models = [m.get('name') for m in resp.json().get('models', []) if m.get('name')] + + ollama_embedding_models = [ + m for m in all_ollama_models if any(k in m for k in ['embed', 'bge', 'embedding']) + ] + generation_models.extend(m for m in all_ollama_models if m not in ollama_embedding_models) + embedding_models.extend(ollama_embedding_models) + except Exception as e: + logger.warning("Could not list Ollama models from %s: %s", host, e) + + # HuggingFace embedding models loaded in-process (see rag_system/main.py EXTERNAL_MODELS). + # harrier-oss-v1-0.6b is the shipped default; the Qwen3 family stays + # selectable for multilingual / long-context corpora. + embedding_models.extend([ + "microsoft/harrier-oss-v1-0.6b", + "Qwen/Qwen3-Embedding-4B", "Qwen/Qwen3-Embedding-0.6B", - "Qwen/Qwen3-Embedding-4B", - "Qwen/Qwen3-Embedding-8B" - ] - embedding_models.extend(huggingface_embedding_models) - - # Sort models for consistent ordering - generation_models.sort() - embedding_models.sort() + "Qwen/Qwen3-Embedding-8B", + ]) self.send_json_response({ - "generation_models": generation_models, - "embedding_models": embedding_models + "generation_models": sorted(set(generation_models)), + "embedding_models": sorted({m for m in embedding_models if m}), }) except Exception as e: self.send_json_response({"error": f"Could not list models: {e}"}, status_code=500) @@ -740,6 +566,7 @@ def send_json_response(self, data, status_code=200): response = json.dumps(data, indent=2) self.wfile.write(response.encode('utf-8')) + def start_server(port=8001): """Starts the API server.""" # Use a reusable TCP server to avoid "address in use" errors on restart @@ -748,10 +575,21 @@ class ReusableTCPServer(socketserver.TCPServer): with ReusableTCPServer(("", port), AdvancedRagApiHandler) as httpd: print(f"๐Ÿš€ Starting Advanced RAG API server on port {port}") + print(f"๐Ÿฉบ Health endpoint: http://localhost:{port}/health") print(f"๐Ÿ’ฌ Chat endpoint: http://localhost:{port}/chat") print(f"โœจ Indexing endpoint: http://localhost:{port}/index") httpd.serve_forever() + if __name__ == "__main__": - # To run this server: python -m rag_system.api_server - start_server() \ No newline at end of file + # To run this server: python -m rag_system.api_server [--port N] + # + # `--port` is parsed here because it was previously *accepted and ignored*: + # `python -m rag_system.api_server --port 8011` silently listened on 8001, + # which is how a test ends up talking to somebody else's server. + import argparse + + _parser = argparse.ArgumentParser(prog="python -m rag_system.api_server", + description="Start the RAG API server.") + _parser.add_argument("--port", type=int, default=8001, help="Port to listen on.") + start_server(port=_parser.parse_args().port) diff --git a/rag_system/api_server_with_progress.py b/rag_system/api_server_with_progress.py deleted file mode 100644 index 438dcb34..00000000 --- a/rag_system/api_server_with_progress.py +++ /dev/null @@ -1,443 +0,0 @@ -import json -import threading -import time -from typing import Dict, List, Any -import logging -from urllib.parse import urlparse, parse_qs -import http.server -import socketserver - -# Import the core logic and batch processing utilities -from rag_system.main import get_agent -from rag_system.utils.batch_processor import ProgressTracker, timer - -# Set up logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -# Global progress tracking storage -ACTIVE_PROGRESS_SESSIONS: Dict[str, Dict[str, Any]] = {} - -# --- Global Singleton for the RAG Agent --- -print("๐Ÿง  Initializing RAG Agent... (This may take a moment)") -RAG_AGENT = get_agent() -if RAG_AGENT is None: - print("โŒ Critical error: RAG Agent could not be initialized. Exiting.") - exit(1) -print("โœ… RAG Agent initialized successfully.") - -class ServerSentEventsHandler: - """Handler for Server-Sent Events (SSE) for real-time progress updates""" - - active_connections: Dict[str, Any] = {} - - @classmethod - def add_connection(cls, session_id: str, response_handler): - """Add a new SSE connection""" - cls.active_connections[session_id] = response_handler - logger.info(f"SSE connection added for session: {session_id}") - - @classmethod - def remove_connection(cls, session_id: str): - """Remove an SSE connection""" - if session_id in cls.active_connections: - del cls.active_connections[session_id] - logger.info(f"SSE connection removed for session: {session_id}") - - @classmethod - def send_event(cls, session_id: str, event_type: str, data: Dict[str, Any]): - """Send an SSE event to a specific session""" - if session_id not in cls.active_connections: - return - - try: - handler = cls.active_connections[session_id] - event_data = json.dumps(data) - message = f"event: {event_type}\ndata: {event_data}\n\n" - handler.wfile.write(message.encode('utf-8')) - handler.wfile.flush() - except Exception as e: - logger.error(f"Failed to send SSE event: {e}") - cls.remove_connection(session_id) - -class RealtimeProgressTracker(ProgressTracker): - """Enhanced ProgressTracker that sends updates via Server-Sent Events""" - - def __init__(self, total_items: int, operation_name: str, session_id: str): - super().__init__(total_items, operation_name) - self.session_id = session_id - self.last_update = 0 - self.update_interval = 1 # Update every 1 second - - # Initialize session progress - ACTIVE_PROGRESS_SESSIONS[session_id] = { - "operation_name": operation_name, - "total_items": total_items, - "processed_items": 0, - "errors_encountered": 0, - "start_time": self.start_time, - "status": "running", - "current_step": "", - "eta_seconds": 0, - "throughput": 0, - "progress_percentage": 0 - } - - # Send initial progress update - self._send_progress_update() - - def update(self, items_processed: int, errors: int = 0, current_step: str = ""): - """Update progress and send notification""" - super().update(items_processed, errors) - - # Update session data - session_data = ACTIVE_PROGRESS_SESSIONS.get(self.session_id) - if session_data: - session_data.update({ - "processed_items": self.processed_items, - "errors_encountered": self.errors_encountered, - "current_step": current_step, - "progress_percentage": (self.processed_items / self.total_items) * 100, - }) - - # Calculate throughput and ETA - elapsed = time.time() - self.start_time - if elapsed > 0: - session_data["throughput"] = self.processed_items / elapsed - remaining = self.total_items - self.processed_items - session_data["eta_seconds"] = remaining / session_data["throughput"] if session_data["throughput"] > 0 else 0 - - # Send update if enough time has passed - current_time = time.time() - if current_time - self.last_update >= self.update_interval: - self._send_progress_update() - self.last_update = current_time - - def finish(self): - """Mark progress as finished and send final update""" - super().finish() - - # Update session status - session_data = ACTIVE_PROGRESS_SESSIONS.get(self.session_id) - if session_data: - session_data.update({ - "status": "completed", - "progress_percentage": 100, - "eta_seconds": 0 - }) - - # Send final update - self._send_progress_update(final=True) - - def _send_progress_update(self, final: bool = False): - """Send progress update via Server-Sent Events""" - session_data = ACTIVE_PROGRESS_SESSIONS.get(self.session_id, {}) - - event_data = { - "session_id": self.session_id, - "progress": session_data.copy(), - "final": final, - "timestamp": time.time() - } - - ServerSentEventsHandler.send_event(self.session_id, "progress", event_data) - -def run_indexing_with_progress(file_paths: List[str], session_id: str): - """Enhanced indexing function with real-time progress tracking""" - from rag_system.pipelines.indexing_pipeline import IndexingPipeline - from rag_system.utils.ollama_client import OllamaClient - import json - - try: - # Send initial status - ServerSentEventsHandler.send_event(session_id, "status", { - "message": "Initializing indexing pipeline...", - "session_id": session_id - }) - - # Load configuration - config_file = "batch_indexing_config.json" - try: - with open(config_file, 'r') as f: - config = json.load(f) - except FileNotFoundError: - # Fallback to default config - config = { - "embedding_model_name": "Qwen/Qwen3-Embedding-0.6B", - "indexing": { - "embedding_batch_size": 50, - "enrichment_batch_size": 10, - "enable_progress_tracking": True - }, - "contextual_enricher": {"enabled": True, "window_size": 1}, - "retrievers": { - "dense": {"enabled": True, "lancedb_table_name": "default_text_table"}, - "bm25": {"enabled": True, "index_name": "default_bm25_index"} - }, - "storage": { - "chunk_store_path": "./index_store/chunks/chunks.pkl", - "lancedb_uri": "./index_store/lancedb", - "bm25_path": "./index_store/bm25" - } - } - - # Initialize components - ollama_client = OllamaClient() - ollama_config = { - "generation_model": "llama3.2:1b", - "embedding_model": "mxbai-embed-large" - } - - # Create enhanced pipeline - pipeline = IndexingPipeline(config, ollama_client, ollama_config) - - # Create progress tracker for the overall process - total_steps = 6 # Rough estimate of pipeline steps - step_tracker = RealtimeProgressTracker(total_steps, "Document Indexing", session_id) - - with timer("Complete Indexing Pipeline"): - try: - # Step 1: Document Processing - step_tracker.update(1, current_step="Processing documents...") - - # Run the indexing pipeline - pipeline.run(file_paths) - - # Update progress through the steps - step_tracker.update(1, current_step="Chunking completed...") - step_tracker.update(1, current_step="BM25 indexing completed...") - step_tracker.update(1, current_step="Contextual enrichment completed...") - step_tracker.update(1, current_step="Vector embeddings completed...") - step_tracker.update(1, current_step="Indexing finalized...") - - step_tracker.finish() - - # Send completion notification - ServerSentEventsHandler.send_event(session_id, "completion", { - "message": f"Successfully indexed {len(file_paths)} file(s)", - "file_count": len(file_paths), - "session_id": session_id - }) - - except Exception as e: - # Send error notification - ServerSentEventsHandler.send_event(session_id, "error", { - "message": str(e), - "session_id": session_id - }) - raise - - except Exception as e: - logger.error(f"Indexing failed for session {session_id}: {e}") - ServerSentEventsHandler.send_event(session_id, "error", { - "message": str(e), - "session_id": session_id - }) - raise - -class EnhancedRagApiHandler(http.server.BaseHTTPRequestHandler): - """Enhanced API handler with progress tracking support""" - - def do_OPTIONS(self): - """Handle CORS preflight requests for frontend integration.""" - self.send_response(200) - self.send_header('Access-Control-Allow-Origin', '*') - self.send_header('Access-Control-Allow-Methods', 'POST, GET, OPTIONS') - self.send_header('Access-Control-Allow-Headers', 'Content-Type') - self.end_headers() - - def do_GET(self): - """Handle GET requests for progress status and SSE streams""" - parsed_path = urlparse(self.path) - - if parsed_path.path == '/progress': - self.handle_progress_status() - elif parsed_path.path == '/stream': - self.handle_progress_stream() - else: - self.send_json_response({"error": "Not Found"}, status_code=404) - - def do_POST(self): - """Handle POST requests for chat and indexing.""" - parsed_path = urlparse(self.path) - - if parsed_path.path == '/chat': - self.handle_chat() - elif parsed_path.path == '/index': - self.handle_index_with_progress() - else: - self.send_json_response({"error": "Not Found"}, status_code=404) - - def handle_chat(self): - """Handles a chat query by calling the agentic RAG pipeline.""" - try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - - query = data.get('query') - if not query: - self.send_json_response({"error": "Query is required"}, status_code=400) - return - - # Use the single, persistent agent instance to run the query - result = RAG_AGENT.run(query) - - # The result is a dict, so we need to dump it to a JSON string - self.send_json_response(result) - - except json.JSONDecodeError: - self.send_json_response({"error": "Invalid JSON"}, status_code=400) - except Exception as e: - self.send_json_response({"error": f"Server error: {str(e)}"}, status_code=500) - - def handle_index_with_progress(self): - """Triggers the document indexing pipeline with real-time progress tracking.""" - try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - - file_paths = data.get('file_paths') - session_id = data.get('session_id') - - if not file_paths or not isinstance(file_paths, list): - self.send_json_response({ - "error": "A 'file_paths' list is required." - }, status_code=400) - return - - if not session_id: - self.send_json_response({ - "error": "A 'session_id' is required for progress tracking." - }, status_code=400) - return - - # Start indexing in a separate thread to avoid blocking - def run_indexing_thread(): - try: - run_indexing_with_progress(file_paths, session_id) - except Exception as e: - logger.error(f"Indexing thread failed: {e}") - - thread = threading.Thread(target=run_indexing_thread) - thread.daemon = True - thread.start() - - # Return immediate response - self.send_json_response({ - "message": f"Indexing started for {len(file_paths)} file(s)", - "session_id": session_id, - "status": "started", - "progress_stream_url": f"http://localhost:8001/stream?session_id={session_id}" - }) - - except json.JSONDecodeError: - self.send_json_response({"error": "Invalid JSON"}, status_code=400) - except Exception as e: - self.send_json_response({"error": f"Failed to start indexing: {str(e)}"}, status_code=500) - - def handle_progress_status(self): - """Handle GET requests for current progress status""" - parsed_url = urlparse(self.path) - params = parse_qs(parsed_url.query) - session_id = params.get('session_id', [None])[0] - - if not session_id: - self.send_json_response({"error": "session_id is required"}, status_code=400) - return - - progress_data = ACTIVE_PROGRESS_SESSIONS.get(session_id) - if not progress_data: - self.send_json_response({"error": "No active progress for this session"}, status_code=404) - return - - self.send_json_response({ - "session_id": session_id, - "progress": progress_data - }) - - def handle_progress_stream(self): - """Handle Server-Sent Events stream for real-time progress""" - parsed_url = urlparse(self.path) - params = parse_qs(parsed_url.query) - session_id = params.get('session_id', [None])[0] - - if not session_id: - self.send_response(400) - self.end_headers() - return - - # Set up SSE headers - self.send_response(200) - self.send_header('Content-Type', 'text/event-stream') - self.send_header('Cache-Control', 'no-cache') - self.send_header('Connection', 'keep-alive') - self.send_header('Access-Control-Allow-Origin', '*') - self.end_headers() - - # Add this connection to the SSE handler - ServerSentEventsHandler.add_connection(session_id, self) - - # Send initial connection message - initial_message = json.dumps({ - "session_id": session_id, - "message": "Progress stream connected", - "timestamp": time.time() - }) - self.wfile.write(f"event: connected\ndata: {initial_message}\n\n".encode('utf-8')) - self.wfile.flush() - - # Keep connection alive - try: - while session_id in ServerSentEventsHandler.active_connections: - time.sleep(1) - # Send heartbeat - heartbeat = json.dumps({"type": "heartbeat", "timestamp": time.time()}) - self.wfile.write(f"event: heartbeat\ndata: {heartbeat}\n\n".encode('utf-8')) - self.wfile.flush() - except Exception as e: - logger.info(f"SSE connection closed for session {session_id}: {e}") - finally: - ServerSentEventsHandler.remove_connection(session_id) - - def send_json_response(self, data, status_code=200): - """Utility to send a JSON response with CORS headers.""" - self.send_response(status_code) - self.send_header('Content-Type', 'application/json') - self.send_header('Access-Control-Allow-Origin', '*') - self.end_headers() - response = json.dumps(data, indent=2) - self.wfile.write(response.encode('utf-8')) - -def start_enhanced_server(port=8000): - """Start the enhanced API server with a reusable TCP socket.""" - - # Use a custom TCPServer that allows address reuse - class ReusableTCPServer(socketserver.TCPServer): - allow_reuse_address = True - - with ReusableTCPServer(("", port), EnhancedRagApiHandler) as httpd: - print(f"๐Ÿš€ Starting Enhanced RAG API server on port {port}") - print(f"๐Ÿ’ฌ Chat endpoint: http://localhost:{port}/chat") - print(f"โœจ Indexing endpoint: http://localhost:{port}/index") - print(f"๐Ÿ“Š Progress endpoint: http://localhost:{port}/progress") - print(f"๐ŸŒŠ Progress stream: http://localhost:{port}/stream") - print(f"๐Ÿ“ˆ Real-time progress tracking enabled via Server-Sent Events!") - httpd.serve_forever() - -if __name__ == '__main__': - # Start the server on a dedicated thread - server_thread = threading.Thread(target=start_enhanced_server) - server_thread.daemon = True - server_thread.start() - - print("๐Ÿš€ Enhanced RAG API server with progress tracking is running.") - print("Press Ctrl+C to stop.") - - # Keep the main thread alive - try: - while True: - time.sleep(1) - except KeyboardInterrupt: - print("\nStopping server...") \ No newline at end of file diff --git a/rag_system/ask_folder.py b/rag_system/ask_folder.py new file mode 100644 index 00000000..6980140e --- /dev/null +++ b/rag_system/ask_folder.py @@ -0,0 +1,250 @@ +"""Ephemeral "ask a folder" mode (roadmap item 4.6). + + python -m rag_system.main ask <folder> "<question>" ["<question>" ...] + +Index a folder into a throwaway LanceDB table, answer, delete the table and +everything else the run created. Nothing is added to the user's real index and +nothing survives the process. + +Why an ephemeral *index* rather than an ephemeral *agent* +--------------------------------------------------------- +The design this borrows from (agentic-file-search) answers folder questions by +letting an agent grep and read files. Our own evidence says not to: filesystem +agents win on small corpora but lose to ranked retrieval as the corpus grows, +at roughly 39x the tokens (BM25-wins-at-scale), and retriever quality dominates +agency (BrowseComp-Plus). So the ephemeral thing here is the *index*, and the +answering path is the one that ships โ€” same chunker, same embedder, same +retriever, same synthesis prompt. There is no second pipeline to keep in sync. + +What is switched off, and why +----------------------------- +The roadmap specifies the ``fast`` profile with no enrichment. On top of that +profile this mode also disables **document overviews**: they cost one LLM call +per document at index time and their only consumers are the agent's triage +router and the (default-off) overview prefilter. Neither runs here โ€” the answer +path skips triage entirely, because someone who typed ``ask <folder> "<q>"`` has +already decided the question is about the folder. + +``--agent`` opts into the full agent loop (decomposition, verification) with +``force_rag`` set. The roadmap says "same pipeline, no agent loop", so that is +the default and the loop is the opt-in. + +Cleanup +------- +Everything this mode writes lives under one ``tempfile.mkdtemp`` directory: the +LanceDB directory and the overview sidecar path both point inside it, so a +single ``rmtree`` in a ``finally`` is the whole teardown. The temp directory is +printed at the start and its removal is reported at the end, so a leak is +visible rather than assumed. ``--keep`` skips the removal for debugging and says +so loudly. + +Indexing and synthesis can take minutes, so the run is very likely to be +interrupted at some point. ``SIGINT`` already unwinds through the ``finally``; +``SIGTERM`` does not, because Python's default handler exits without unwinding โ€” +so it is temporarily rebound to raise ``SystemExit``. Both signals therefore +delete the temp index. ``SIGKILL`` cannot be caught and will leak the directory; +that is a property of ``kill -9``, not something this module can fix, and the +leaked directory is a ``localgpt-ask-*`` under ``$TMPDIR``. +""" + +from __future__ import annotations + +import contextlib +import os +import shutil +import signal +import sys +import tempfile +import time +import uuid +from typing import Any, Dict, List, Sequence + +# How much of a cited chunk to echo under the answer. +_CITATION_PREVIEW_CHARS = 220 + +# Citations printed per answer. The point is traceability, not a second copy of +# the corpus on the terminal. +_MAX_CITATIONS = 5 + + +def _ephemeral_config(mode: str, temp_dir: str, table_name: str) -> Dict[str, Any]: + """The chosen profile, redirected wholly inside *temp_dir*.""" + from rag_system.factory import get_pipeline_config + + config = get_pipeline_config(mode) + config["storage"]["lancedb_uri"] = os.path.join(temp_dir, "lancedb") + config["storage"]["db_path"] = os.path.join(temp_dir, "lancedb") + config["storage"]["text_table_name"] = table_name + + # No enrichment (roadmap 4.6) โ€” an LLM call per chunk is exactly what a + # throwaway index should not pay for. + config["contextual_enricher"] = {"enabled": False, "window_size": 1} + # No overviews: an LLM call per document whose only consumers do not run here. + config["overview"] = {"enabled": False} + # Belt and braces โ€” if a future profile turns overviews back on, the JSONL + # and its vector sidecar still land in the temp directory rather than in the + # user's index_store/. + config["overview_path"] = os.path.join(temp_dir, "overviews", f"{table_name}.jsonl") + # A second embedding pass over every document, for a corpus we are about to + # delete. + config.setdefault("retrieval", {})["latechunk"] = {"enabled": False} + return config + + +_UNSET = object() + + +def _sigterm_raises() -> Any: + """Make SIGTERM raise ``SystemExit`` so ``finally`` blocks actually run. + + Python's default SIGTERM disposition exits the process *without* unwinding, + so a ``kill`` (or a harness timeout) during a long index build would leave + the temp directory behind. Returns the previous handler for + ``_restore_sigterm``, or ``_UNSET`` when the swap was not possible (not the + main thread, or a platform without SIGTERM). + """ + def _raise(_signum, _frame): + raise SystemExit(143) + + try: + previous = signal.getsignal(signal.SIGTERM) + signal.signal(signal.SIGTERM, _raise) + return previous + except (ValueError, OSError, AttributeError): + return _UNSET + + +def _restore_sigterm(previous: Any) -> None: + if previous is _UNSET: + return + with contextlib.suppress(Exception): + signal.signal(signal.SIGTERM, previous) + + +def _print_citations(source_documents: Sequence[Dict[str, Any]], out) -> None: + if not source_documents: + out(" (no sources)") + return + for i, doc in enumerate(source_documents[:_MAX_CITATIONS], start=1): + text = (doc.get("text") or "").strip().replace("\n", " ") + if len(text) > _CITATION_PREVIEW_CHARS: + text = text[:_CITATION_PREVIEW_CHARS] + "โ€ฆ" + score = doc.get("rerank_score", doc.get("score")) + score_str = f"{score:.4f}" if isinstance(score, (int, float)) else str(score) + out(f" [{i}] {doc.get('document_id')}#{doc.get('chunk_index')} " + f"(score {score_str})") + out(f" {text}") + extra = len(source_documents) - _MAX_CITATIONS + if extra > 0: + out(f" โ€ฆ and {extra} more source chunk(s)") + + +def _answer(agent, question: str, table_name: str, use_agent: bool, + filters: Any, out) -> Dict[str, Any]: + """One question, through either the retrieval pipeline or the whole agent.""" + if use_agent: + # force_rag: the user pointed at a folder, so triage has nothing to decide. + return agent.run(question, table_name=table_name, force_rag=True, + filters=filters) + return agent.retrieval_pipeline.run(question, table_name, filters=filters) + + +def ask_folder(path: str, questions: Sequence[str], *, mode: str = "fast", + interactive: bool = False, use_agent: bool = False, + filters: Any = None, keep: bool = False, out=print) -> int: + """Index *path*, answer *questions*, delete the index. Returns an exit code.""" + from rag_system.agent.loop import Agent + from rag_system.factory import _build_llm_client + from rag_system.main import SUPPORTED_DOCUMENT_EXTENSIONS, _collect_file_paths + from rag_system.pipelines.indexing_pipeline import IndexingPipeline + + try: + file_paths = _collect_file_paths(path) + except FileNotFoundError as e: + out(f"โŒ {e}") + return 1 + if not file_paths: + out(f"โŒ No indexable documents in {path} " + f"(supported: {', '.join(SUPPORTED_DOCUMENT_EXTENSIONS)}).") + return 1 + + if not questions and not interactive: + out("โŒ Nothing to ask. Pass one or more questions, or use --interactive.") + return 1 + + # A fresh table name per run, so two concurrent `ask` runs cannot collide + # even if they somehow shared a directory. + table_name = f"ask_{uuid.uuid4().hex[:12]}" + temp_dir = tempfile.mkdtemp(prefix="localgpt-ask-") + started = time.time() + + out(f"๐Ÿ“‚ ask: {len(file_paths)} file(s) from {os.path.abspath(path)}") + out(f"๐Ÿ—‘๏ธ ephemeral index: table '{table_name}' in {temp_dir} (profile '{mode}')") + + exit_code = 0 + previous_sigterm = _sigterm_raises() + try: + config = _ephemeral_config(mode, temp_dir, table_name) + llm_client, llm_config = _build_llm_client() + + index_started = time.time() + IndexingPipeline(config, llm_client, llm_config).run(file_paths) + out(f"โฑ๏ธ indexed in {time.time() - index_started:.1f}s") + + # The standard agent, on the ephemeral config. `use_agent` picks which of + # its two entry points answers: the retrieval pipeline (the roadmap's + # "same pipeline, no agent loop") or the full loop. + agent = Agent(pipeline_configs=config, llm_client=llm_client, + ollama_config=llm_config) + + pending: List[str] = list(questions) + asked = 0 + while True: + if not pending: + if not interactive: + break + try: + follow_up = input("\nโ“ Ask a follow-up (blank line to finish): ").strip() + except EOFError: + out("") + break + if not follow_up: + break + pending.append(follow_up) + + question = pending.pop(0) + asked += 1 + out(f"\n{'=' * 72}\nโ“ {question}\n{'=' * 72}") + answer_started = time.time() + result = _answer(agent, question, table_name, use_agent, filters, out) + out(f"\n๐Ÿ’ฌ {result.get('answer', '').strip()}") + out(f"\n๐Ÿ“Ž Sources ({time.time() - answer_started:.1f}s):") + _print_citations(result.get("source_documents") or [], out) + + if asked == 0: + out("โ„น๏ธ No questions asked.") + + except Exception as e: + out(f"โŒ ask failed: {type(e).__name__}: {e}") + exit_code = 1 + finally: + _restore_sigterm(previous_sigterm) + # The teardown is the feature. Everything this run wrote is under + # temp_dir, so one rmtree is the whole of it โ€” but say what happened + # either way, because "assume it cleaned up" is how leaks survive. + if keep: + out(f"\nโš ๏ธ --keep: the ephemeral index was NOT deleted. Remove it yourself: {temp_dir}") + else: + shutil.rmtree(temp_dir, ignore_errors=True) + if os.path.exists(temp_dir): + out(f"\nโŒ Could not remove the ephemeral index at {temp_dir}") + exit_code = 1 + else: + out(f"\n๐Ÿงน Removed the ephemeral index ({temp_dir}).") + out(f"โฑ๏ธ total {time.time() - started:.1f}s") + + return exit_code + + +if __name__ == "__main__": # pragma: no cover - the real entry point is main.py + sys.exit(ask_folder(sys.argv[1], sys.argv[2:])) diff --git a/rag_system/factory.py b/rag_system/factory.py index 77a79e89..6e2e2e62 100644 --- a/rag_system/factory.py +++ b/rag_system/factory.py @@ -1,82 +1,66 @@ +import copy + from dotenv import load_dotenv -def get_agent(mode: str = "default"): - """ - Factory function to get an instance of the RAG agent based on the specified mode. - This uses local imports to prevent circular dependencies. + +def _build_llm_client(): + """Create the LLM client and its config for the active LLM_BACKEND. + + Uses local imports to prevent circular dependencies with rag_system.main. """ - from rag_system.agent.loop import Agent - from rag_system.utils.ollama_client import OllamaClient - from rag_system.main import PIPELINE_CONFIGS, OLLAMA_CONFIG, LLM_BACKEND, WATSONX_CONFIG + from rag_system.main import LLM_BACKEND, OLLAMA_CONFIG, WATSONX_CONFIG - load_dotenv() - - # Initialize the appropriate LLM client based on backend configuration if LLM_BACKEND.lower() == "watsonx": from rag_system.utils.watsonx_client import WatsonXClient - + if not WATSONX_CONFIG["api_key"] or not WATSONX_CONFIG["project_id"]: raise ValueError( "Watson X configuration incomplete. Please set WATSONX_API_KEY and WATSONX_PROJECT_ID " "environment variables." ) - - llm_client = WatsonXClient( + + client = WatsonXClient( api_key=WATSONX_CONFIG["api_key"], project_id=WATSONX_CONFIG["project_id"], - url=WATSONX_CONFIG["url"] + url=WATSONX_CONFIG["url"], ) - llm_config = WATSONX_CONFIG - else: - llm_client = OllamaClient(host=OLLAMA_CONFIG["host"]) - llm_config = OLLAMA_CONFIG - - config = PIPELINE_CONFIGS.get(mode, PIPELINE_CONFIGS['default']) - - if 'storage' not in config: - config['storage'] = { - 'db_path': 'lancedb', - 'text_table_name': 'text_pages_default', - 'image_table_name': 'image_pages' - } - - agent = Agent( - pipeline_configs=config, - llm_client=llm_client, - ollama_config=llm_config + return client, WATSONX_CONFIG + + from rag_system.utils.ollama_client import OllamaClient + + return OllamaClient(host=OLLAMA_CONFIG["host"]), OLLAMA_CONFIG + + +def get_pipeline_config(mode: str = "default") -> dict: + """Return a deep copy of a pipeline profile so callers cannot mutate the master config.""" + from rag_system.main import PIPELINE_CONFIGS + + return copy.deepcopy(PIPELINE_CONFIGS.get(mode, PIPELINE_CONFIGS["default"])) + + +def get_agent(mode: str = "default"): + """Factory function to get an instance of the RAG agent for the specified mode.""" + from rag_system.agent.loop import Agent + + load_dotenv() + + llm_client, llm_config = _build_llm_client() + config = get_pipeline_config(mode) + + return Agent( + pipeline_configs=config, + llm_client=llm_client, + ollama_config=llm_config, ) - return agent + def get_indexing_pipeline(mode: str = "default"): - """ - Factory function to get an instance of the Indexing Pipeline. - """ + """Factory function to get an instance of the Indexing Pipeline for the specified mode.""" from rag_system.pipelines.indexing_pipeline import IndexingPipeline - from rag_system.main import PIPELINE_CONFIGS, OLLAMA_CONFIG, LLM_BACKEND, WATSONX_CONFIG - from rag_system.utils.ollama_client import OllamaClient load_dotenv() - - # Initialize the appropriate LLM client based on backend configuration - if LLM_BACKEND.lower() == "watsonx": - from rag_system.utils.watsonx_client import WatsonXClient - - if not WATSONX_CONFIG["api_key"] or not WATSONX_CONFIG["project_id"]: - raise ValueError( - "Watson X configuration incomplete. Please set WATSONX_API_KEY and WATSONX_PROJECT_ID " - "environment variables." - ) - - llm_client = WatsonXClient( - api_key=WATSONX_CONFIG["api_key"], - project_id=WATSONX_CONFIG["project_id"], - url=WATSONX_CONFIG["url"] - ) - llm_config = WATSONX_CONFIG - else: - llm_client = OllamaClient(host=OLLAMA_CONFIG["host"]) - llm_config = OLLAMA_CONFIG - - config = PIPELINE_CONFIGS.get(mode, PIPELINE_CONFIGS['default']) - - return IndexingPipeline(config, llm_client, llm_config) \ No newline at end of file + + llm_client, llm_config = _build_llm_client() + config = get_pipeline_config(mode) + + return IndexingPipeline(config, llm_client, llm_config) diff --git a/rag_system/indexing/contextualizer.py b/rag_system/indexing/contextualizer.py index 714c65d3..5228c01f 100644 --- a/rag_system/indexing/contextualizer.py +++ b/rag_system/indexing/contextualizer.py @@ -141,41 +141,4 @@ def process_chunk_batch(chunk_indices): "Contextual Enrichment" ) - return enriched_chunks - - def enrich_chunks_sequential(self, chunks: List[Dict[str, Any]], window_size: int = 1) -> List[Dict[str, Any]]: - """Sequential enrichment method (legacy) - kept for comparison""" - if not chunks: - return [] - - logger.info(f"Enriching {len(chunks)} chunks sequentially (window_size={window_size})...") - enriched_chunks = [] - - for i, chunk in enumerate(chunks): - local_context_text = create_contextual_window(chunks, chunk_index=i, window_size=window_size) - - # The summary is generated based on the original, unmodified text - original_text = chunk['text'] - summary = self._generate_summary(local_context_text, original_text) - - new_chunk = chunk.copy() - - # Ensure metadata is a dictionary - if 'metadata' not in new_chunk or not isinstance(new_chunk['metadata'], dict): - new_chunk['metadata'] = {} - - # Store original text and summary in metadata - new_chunk['metadata']['original_text'] = original_text - new_chunk['metadata']['contextual_summary'] = "N/A" - - # Prepend the context summary ONLY if it was successfully generated - if summary: - new_chunk['text'] = f"Context: {summary}\n\n---\n\n{original_text}" - new_chunk['metadata']['contextual_summary'] = summary - - enriched_chunks.append(new_chunk) - - if (i + 1) % 10 == 0 or i == len(chunks) - 1: - logger.info(f" ...processed {i+1}/{len(chunks)} chunks.") - return enriched_chunks \ No newline at end of file diff --git a/rag_system/indexing/crossref.py b/rag_system/indexing/crossref.py new file mode 100644 index 00000000..22dc29e7 --- /dev/null +++ b/rag_system/indexing/crossref.py @@ -0,0 +1,244 @@ +"""Index-time extraction of intra-corpus cross-references (roadmap item 4.2). + +Embeddings cannot follow a pointer. A chunk that says *"the fee schedule is set +out in Exhibit B"* is a perfect lexical and semantic match for a question about +fees, but the fee *numbers* live in a different document that shares almost no +vocabulary with the query. No amount of reranking recovers that document, +because it was never a candidate. + +This module extracts those pointers with deterministic regexes at index time and +stores them on the chunk as ``metadata.crossrefs``:: + + [{"kind": "exhibit" | "section" | "document", + "ref": "<normalized reference>", + "target_doc": "<document_id>" | None}] + +Three deliberate limits: + +* **No LLM.** Extraction is a handful of regexes over text we already have in + memory, so it is free and can be on by default. The *query-time hop* that acts + on this metadata is a separate, default-off flag. +* **Resolution is name-based only.** A reference resolves when some other + document being indexed has a filename or title that contains it + ("Exhibit B" -> ``exhibit_b.pdf``). Anything else stays ``target_doc: None``: + the reference is still recorded (it is true, and a UI can show it), it just + has nowhere to hop to. +* **Never resolves to the chunk's own document.** A contract that mentions + "Exhibit B" fifty times inside ``exhibit_b.pdf`` produces no useful hop, and a + document that repeats its own title would otherwise self-resolve on every + chunk. +""" + +from __future__ import annotations + +import os +import re +from typing import Any, Dict, Iterable, List, Optional, Sequence + +# A chunk that carries more than this many distinct references is almost always +# a table of contents or an index page; keeping the first few is enough to hop. +MAX_CROSSREFS_PER_CHUNK = 8 + +# A document-name mention has to clear this bar before it counts, otherwise +# short generic filenames ("api.md", "v2.pdf") match half the corpus. +_MIN_NAME_CHARS = 6 +_MIN_NAME_TOKENS = 2 + +# Label families that behave like "Exhibit B": a label word plus a letter or a +# dotted number. All of them are recorded under kind "exhibit"; the label word +# itself survives in ``ref`` ("schedule 2.1"), so nothing is conflated. +_LABEL_WORDS = ("exhibit", "appendix", "schedule", "annex", "attachment", "addendum") + +# The label word is case-insensitive; the single-letter form is NOT, because a +# case-insensitive ``[A-Z]`` turns "appendix in the margin" into "appendix i". +_LABEL_RE = re.compile( + r"\b(?i:(" + "|".join(_LABEL_WORDS) + r"))" + r"[\sย ]+(?i:no\.?[\sย ]*|#[\sย ]*)?" + r"([A-Z](?![A-Za-z])|\d+(?:\.\d+)*)" +) + +_SECTION_RE = re.compile( + r"\b(?i:(section|clause|article))[\sย ]+(\d+(?:\.\d+)*)\b" +) + +_SECTION_SYMBOL_RE = re.compile(r"ยง[\sย ]*(\d+(?:\.\d+)*)") + +_EXTENSION_RE = re.compile(r"\.[A-Za-z0-9]{1,5}$") +_NON_ALNUM_RE = re.compile(r"[^0-9A-Za-z]+") + + +def normalize_name(value: str) -> str: + """A document id, filename or title reduced to space-separated lowercase words. + + ``"Exhibit_B.pdf"`` and ``"exhibit b"`` both become ``"exhibit b"``, which is + what makes a textual reference comparable with a filename. + """ + if not value: + return "" + base = os.path.basename(str(value)) + base = _EXTENSION_RE.sub("", base) + return _NON_ALNUM_RE.sub(" ", base).strip().lower() + + +def _lookup_key(ref: str) -> str: + """``"section 4.3"`` -> ``"section 4 3"`` so it can be matched against names.""" + return _NON_ALNUM_RE.sub(" ", (ref or "").lower()).strip() + + +def _normalize_text(text: str) -> str: + return _NON_ALNUM_RE.sub(" ", (text or "")).lower() + + +class CrossRefExtractor: + """Extracts and resolves references against a known set of document ids. + + *known_documents* is the id set the references are resolved against: the + documents in the current indexing batch, plus (best effort) whatever is + already in the target table, so an incremental add can still point at an + earlier one. + """ + + def __init__(self, known_documents: Sequence[str] = ()): + # normalized name -> document id. Later duplicates lose, which is + # arbitrary but stable, and duplicate normalized filenames in one corpus + # are already ambiguous for a human. + self._by_name: Dict[str, str] = {} + for doc_id in known_documents or (): + name = normalize_name(doc_id) + if name: + self._by_name.setdefault(name, doc_id) + # Corpora ordered with numeric filename prefixes ("05_escrow_ + # agreement.pdf") are referenced in prose by title ("the Escrow + # Agreement"), never by prefix, so index the stripped form too. + # The full-name entry above still wins ties. + stripped = re.sub(r"^\d+\s+", "", name) + if stripped != name and ( + len(stripped) >= _MIN_NAME_CHARS or len(stripped.split()) >= _MIN_NAME_TOKENS + ): + self._by_name.setdefault(stripped, doc_id) + # Longest names first so "northwind leave policy" wins over "policy". + self._mention_names = sorted( + ( + name for name in self._by_name + if len(name) >= _MIN_NAME_CHARS or len(name.split()) >= _MIN_NAME_TOKENS + ), + key=len, + reverse=True, + ) + self._name_patterns = { + name: re.compile(r"\b" + re.escape(name) + r"\b") for name in self._mention_names + } + # Cheap pre-computed word-boundary matchers for reference resolution. + self._resolve_cache: Dict[str, Optional[str]] = {} + + # -- resolution -------------------------------------------------------- + + def _resolve(self, ref: str, self_doc_id: Optional[str]) -> Optional[str]: + key = _lookup_key(ref) + if not key: + return None + cached = self._resolve_cache.get(key, "__miss__") + if cached == "__miss__": + pattern = re.compile(r"\b" + re.escape(key) + r"\b") + hit = None + # Prefer an exact name match, then a containing name. + if key in self._by_name: + hit = self._by_name[key] + else: + for name in sorted(self._by_name, key=len): + if pattern.search(name): + hit = self._by_name[name] + break + self._resolve_cache[key] = hit + cached = hit + if cached is not None and self_doc_id is not None and cached == self_doc_id: + return None + return cached + + # -- extraction -------------------------------------------------------- + + def extract(self, text: str, self_doc_id: Optional[str] = None) -> List[Dict[str, Any]]: + """Every distinct reference in *text*, in order of first appearance.""" + if not text: + return [] + + found: List[Dict[str, Any]] = [] + seen = set() + seen_refs = set() + + def add(kind: str, ref: str, target: Optional[str]) -> None: + item = (kind, ref) + if item in seen: + return + seen.add(item) + seen_refs.add(ref) + found.append({"kind": kind, "ref": ref, "target_doc": target}) + + for match in _LABEL_RE.finditer(text): + label = match.group(1).lower() + value = match.group(2) + ref = f"{label} {value.lower()}" + add("exhibit", ref, self._resolve(ref, self_doc_id)) + + for match in _SECTION_RE.finditer(text): + ref = f"section {match.group(2)}" + add("section", ref, self._resolve(ref, self_doc_id)) + + for match in _SECTION_SYMBOL_RE.finditer(text): + ref = f"section {match.group(1)}" + add("section", ref, self._resolve(ref, self_doc_id)) + + if self._mention_names: + haystack = _normalize_text(text) + self_name = normalize_name(self_doc_id) if self_doc_id else "" + for name in self._mention_names: + if name == self_name or name in seen_refs: + # Already recorded by the label pass ("Exhibit B" in a corpus + # that also has exhibit_b.pdf) โ€” one record, not two. + continue + target = self._by_name.get(name) + if target is not None and target == self_doc_id: + continue + if self._name_patterns[name].search(haystack): + add("document", name, target) + + return found[:MAX_CROSSREFS_PER_CHUNK] + + +def annotate_chunks( + doc_chunks: Dict[str, List[Dict[str, Any]]], + known_documents: Optional[Iterable[str]] = None, +) -> Dict[str, int]: + """Stamp ``metadata.crossrefs`` on every chunk. Returns a small stats dict. + + *doc_chunks* maps document id -> that document's chunks (the shape + ``IndexingPipeline`` already keeps for late chunking). Chunks are mutated in + place; chunks with no references get no key at all, so the stored metadata + does not grow for corpora that have none. + """ + ids = list(doc_chunks.keys()) + if known_documents: + for extra in known_documents: + if extra not in ids: + ids.append(extra) + + extractor = CrossRefExtractor(ids) + stats = {"chunks_with_refs": 0, "refs": 0, "resolved": 0, "documents_linked": 0} + linked_targets = set() + + for doc_id, chunks in doc_chunks.items(): + for chunk in chunks: + text = chunk.get("text") or "" + refs = extractor.extract(text, self_doc_id=doc_id) + if not refs: + continue + chunk.setdefault("metadata", {})["crossrefs"] = refs + stats["chunks_with_refs"] += 1 + stats["refs"] += len(refs) + for ref in refs: + if ref["target_doc"]: + stats["resolved"] += 1 + linked_targets.add(ref["target_doc"]) + + stats["documents_linked"] = len(linked_targets) + return stats diff --git a/rag_system/indexing/embedders.py b/rag_system/indexing/embedders.py index b48648f2..01ab5af9 100644 --- a/rag_system/indexing/embedders.py +++ b/rag_system/indexing/embedders.py @@ -1,10 +1,134 @@ -# from rag_system.indexing.representations import BM25Generator import lancedb +import os import pyarrow as pa -from typing import List, Dict, Any +from typing import Any, Dict, List, Optional import numpy as np import json +# --------------------------------------------------------------------------- +# Per-table embedder identity + vector-normalization marker +# --------------------------------------------------------------------------- +# Two facts have to travel with a LanceDB table, because neither can be +# recovered from the vectors themselves: +# +# 1. WHICH embedding model wrote them. The vector-width check below cannot +# catch a swap between two same-width models (harrier-oss-v1-0.6b and +# Qwen3-Embedding-0.6B are both 1024-dim), and appending one model's +# vectors to the other's table silently produces nonsense rankings. +# 2. WHETHER they are L2-normalized. Both model cards specify cosine +# similarity, but LanceDB's default metric is L2; L2 ordering equals +# cosine ordering only when every vector is unit length. Normalizing +# invalidates vectors written before this existed, so it is recorded +# per table rather than assumed globally. +# +# Primary store: Arrow schema metadata on the table, which lancedb 0.36.0 +# round-trips through create_table/open_table (verified on this tree). If a +# LanceDB version ever drops it, a sidecar JSON is written next to the database +# instead โ€” under <db_path>/table_meta/<table>.json, *not* a single global +# directory, because different indexes legitimately reuse the same table name +# in different database directories (the eval harness does exactly that). +# +# A table with neither marker is a legacy table: its embedder is unknown and +# its vectors are unnormalized. It keeps working, unnormalized, with a warning. + +_META_MODEL_KEY = b"localgpt_embedding_model" +_META_NORMALIZED_KEY = b"localgpt_normalized" +_SIDECAR_DIRNAME = "table_meta" + + +class EmbedderMismatchError(RuntimeError): + """A table was written by a different embedding model than the configured one.""" + + +def l2_normalize(vector: np.ndarray) -> np.ndarray: + """Unit-length copy of *vector*; returned unchanged when that is impossible.""" + array = np.asarray(vector, dtype=np.float32) + if not np.isfinite(array).all(): + # Leave it alone so the NaN/Inf reporting downstream stays accurate. + return array + norm = float(np.linalg.norm(array)) + if norm == 0.0: + return array + return array / norm + + +def _sidecar_path(db_path: str, table_name: str) -> str: + return os.path.join(db_path, _SIDECAR_DIRNAME, f"{table_name}.json") + + +def _read_sidecar(db_path: Optional[str], table_name: Optional[str]) -> Optional[Dict[str, Any]]: + if not db_path or not table_name: + return None + path = _sidecar_path(db_path, table_name) + try: + with open(path, "r", encoding="utf-8") as fh: + return json.load(fh) + except (OSError, ValueError): + return None + + +def _write_sidecar(db_path: str, table_name: str, model_name: str, normalized: bool) -> None: + path = _sidecar_path(db_path, table_name) + try: + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "w", encoding="utf-8") as fh: + json.dump({"embedding_model": model_name, "normalized": bool(normalized)}, fh, indent=2) + except OSError as e: + print(f"โš ๏ธ Could not write table marker {path}: {e}") + + +def table_schema_metadata(model_name: str, normalized: bool) -> Dict[bytes, bytes]: + return { + _META_MODEL_KEY: model_name.encode("utf-8"), + _META_NORMALIZED_KEY: (b"true" if normalized else b"false"), + } + + +def read_table_marker(tbl, db_path: Optional[str] = None, + table_name: Optional[str] = None) -> Optional[Dict[str, Any]]: + """The embedder identity recorded for *tbl*, or None for a legacy table.""" + metadata = getattr(getattr(tbl, "schema", None), "metadata", None) or {} + raw_model = metadata.get(_META_MODEL_KEY) + if raw_model: + raw_norm = metadata.get(_META_NORMALIZED_KEY, b"false") + return { + "embedding_model": raw_model.decode("utf-8"), + "normalized": raw_norm.decode("utf-8").lower() == "true", + "source": "lancedb schema metadata", + } + sidecar = _read_sidecar(db_path, table_name) + if sidecar and sidecar.get("embedding_model"): + return { + "embedding_model": sidecar["embedding_model"], + "normalized": bool(sidecar.get("normalized")), + "source": _sidecar_path(db_path, table_name), + } + return None + + +def assert_embedder_matches(table_name: str, marker: Dict[str, Any], configured_model: str) -> None: + """Raise when *marker* names a different embedder than *configured_model*.""" + recorded = marker.get("embedding_model") + if not recorded or not configured_model or recorded == configured_model: + return + raise EmbedderMismatchError( + f"Table '{table_name}' was built with embedding model '{recorded}' but the " + f"pipeline is configured for '{configured_model}'. The two produce vectors in " + f"different spaces (a matching vector width does not make them compatible), so " + f"any result from this table would be meaningless. Rebuild the index with " + f"'{configured_model}', or set EMBEDDING_MODEL='{recorded}' to keep using it." + ) + + +def legacy_table_warning(table_name: str) -> str: + return ( + f"โš ๏ธ Table '{table_name}' carries no embedder marker โ€” it was built before " + f"localGPT recorded one. Its embedding model cannot be verified and its vectors " + f"are unnormalized, so scores use legacy unnormalized vectors; a rebuilt index " + f"is recommended." + ) + + class LanceDBManager: def __init__(self, db_path: str): self.db_path = db_path @@ -27,15 +151,59 @@ class VectorIndexer: def __init__(self, db_manager: LanceDBManager): self.db_manager = db_manager - def index(self, table_name: str, chunks: List[Dict[str, Any]], embeddings: np.ndarray): + @staticmethod + def _table_vector_dim(tbl) -> int | None: + """Vector width of an existing LanceDB table, or None if it can't be read.""" + try: + field = tbl.schema.field("vector") + except (KeyError, AttributeError): + return None + return getattr(field.type, "list_size", None) + + def index(self, table_name: str, chunks: List[Dict[str, Any]], embeddings: np.ndarray, + embedding_model: Optional[str] = None): if len(chunks) != len(embeddings): raise ValueError("The number of chunks and embeddings must be the same.") if not chunks: print("No chunks to index.") return - vector_dim = embeddings[0].shape[0] - + # Dimensionality always comes from the vectors the loaded model produced. + vector_dim = int(embeddings[0].shape[0]) + + db = self.db_manager.db # underlying LanceDB connection + db_path = getattr(self.db_manager, "db_path", None) + table_exists = bool(hasattr(db, "table_names") and table_name in db.table_names()) + + # ------------------------------------------------------------------ + # Decide, before touching the vectors, which table this is: + # new table -> record the embedder, write normalized vectors + # marked table -> embedder must match; follow the table's own flag + # legacy table -> unknown embedder, unnormalized; warn and comply + # ------------------------------------------------------------------ + existing_tbl = None + normalize = bool(embedding_model) # can't claim an identity we weren't given + if table_exists: + existing_tbl = self.db_manager.get_table(table_name) + existing_dim = self._table_vector_dim(existing_tbl) + if existing_dim is not None and existing_dim != vector_dim: + raise ValueError( + f"Table '{table_name}' stores {existing_dim}-dim vectors but the current " + f"embedding model produced {vector_dim}-dim vectors. Changing the embedding " + f"model requires rebuilding the index." + ) + marker = read_table_marker(existing_tbl, db_path, table_name) + if marker is None: + print(legacy_table_warning(table_name)) + normalize = False + else: + if embedding_model: + assert_embedder_matches(table_name, marker, embedding_model) + normalize = bool(marker["normalized"]) + + if normalize: + embeddings = [l2_normalize(v) for v in embeddings] + # The schema stores the text that was used for the embedding (potentially enriched) # and the full metadata object as a JSON string. schema = pa.schema([ @@ -45,7 +213,7 @@ def index(self, table_name: str, chunks: List[Dict[str, Any]], embeddings: np.nd pa.field("document_id", pa.string()), pa.field("chunk_index", pa.int32()), pa.field("metadata", pa.string()) - ]) + ], metadata=table_schema_metadata(embedding_model, normalize) if embedding_model else None) data = [] skipped_count = 0 @@ -93,14 +261,35 @@ def index(self, table_name: str, chunks: List[Dict[str, Any]], embeddings: np.nd return # Incremental indexing: append to existing table if present, otherwise create it - db = self.db_manager.db # underlying LanceDB connection - - if hasattr(db, "table_names") and table_name in db.table_names(): - tbl = self.db_manager.get_table(table_name) + if existing_tbl is not None: + tbl = existing_tbl + # Re-indexing must replace, not duplicate: drop the rows a previous + # build wrote for these document_ids before appending the new ones. + doc_ids = sorted({row["document_id"] for row in data}) + for doc_id in doc_ids: + if "'" in doc_id: + # Refuse, don't escape (see rag_system/retrieval/filters.py). + raise ValueError( + f"Refusing to delete rows for document_id containing a single quote: {doc_id!r}" + ) + if doc_ids: + quoted = ", ".join(f"'{d}'" for d in doc_ids) + result = tbl.delete(f"document_id IN ({quoted})") + deleted = getattr(result, "num_deleted_rows", None) + if deleted: + print(f"๐Ÿ—‘๏ธ Removed {deleted} stale row(s) for {len(doc_ids)} document(s) from '{table_name}'.") print(f"Appending {len(data)} vectors to existing table '{table_name}'.") else: print(f"Creating table '{table_name}' (new) and adding {len(data)} vectors...") tbl = self.db_manager.create_table(table_name, schema=schema, mode="create") + if embedding_model: + # Trust nothing: re-read the marker off the created table. If this + # LanceDB build dropped the Arrow schema metadata, fall back to the + # sidecar so the guard still has something to compare against. + if read_table_marker(self.db_manager.get_table(table_name)) is None and db_path: + _write_sidecar(db_path, table_name, embedding_model, normalize) + print(f"๐Ÿ”– Table '{table_name}' marked: embedder='{embedding_model}', " + f"normalized={str(normalize).lower()}.") # Add data with NaN handling configuration try: @@ -117,10 +306,6 @@ def index(self, table_name: str, chunks: List[Dict[str, Any]], embeddings: np.nd print(f"โŒ Failed to add data even with NaN fill: {e2}") raise -# BM25Indexer is no longer needed as we are moving to LanceDB's native FTS. -# class BM25Indexer: -# ... - if __name__ == '__main__': print("embedders.py updated for contextual enrichment.") diff --git a/rag_system/indexing/graph_extractor.py b/rag_system/indexing/graph_extractor.py deleted file mode 100644 index 70084a44..00000000 --- a/rag_system/indexing/graph_extractor.py +++ /dev/null @@ -1,86 +0,0 @@ -from typing import List, Dict, Any -import json -from rag_system.utils.ollama_client import OllamaClient - -class GraphExtractor: - """ - Extracts entities and relationships from text chunks using a live Ollama model. - """ - def __init__(self, llm_client: OllamaClient, llm_model: str): - self.llm_client = llm_client - self.llm_model = llm_model - print(f"Initialized GraphExtractor with Ollama model '{self.llm_model}'.") - - def extract(self, chunks: List[Dict[str, Any]]) -> Dict[str, List[Dict]]: - all_entities = {} - all_relationships = set() - - print(f"Extracting graph from {len(chunks)} chunks with Ollama...") - for i, chunk in enumerate(chunks): - # Step 1: Extract Entities - entity_prompt = f""" - From the following text, extract key entities (people, companies, locations). - Return the answer as a JSON object with a single key 'entities', which is a list of strings. - Each entity should be a short, specific name, not a long string of text. - - Text: "{chunk['text']}" - """ - - entity_response = self.llm_client.generate_completion( - self.llm_model, - entity_prompt, - format="json" - ) - - entity_response_text = entity_response.get('response', '{}') - - try: - entity_data = json.loads(entity_response_text) - entities = entity_data.get('entities', []) - - if not entities: - continue - - # Clean up entities - cleaned_entities = [] - for entity in entities: - if len(entity) < 50 and not any(c in entity for c in "[]{}()"): - cleaned_entities.append(entity) - - if not cleaned_entities: - continue - - # Step 2: Extract Relationships - relationship_prompt = f""" - Given the following entities: {cleaned_entities} - And the following text: "{chunk['text']}" - Extract the relationships between the entities. - Return the answer as a JSON object with a single key 'relationships', which is a list of objects, each with 'source', 'target', and 'label'. - """ - - relationship_response = self.llm_client.generate_completion( - self.llm_model, - relationship_prompt, - format="json" - ) - - relationship_response_text = relationship_response.get('response', '{}') - relationship_data = json.loads(relationship_response_text) - - for entity_name in cleaned_entities: - all_entities[entity_name] = {"id": entity_name, "type": "Unknown"} # Placeholder type - - for rel in relationship_data.get("relationships", []): - if 'source' in rel and 'target' in rel and 'label' in rel: - all_relationships.add( - (rel['source'], rel['target'], rel['label']) - ) - - except json.JSONDecodeError: - print(f"Warning: Could not decode JSON from LLM for chunk {i+1}.") - continue - - return { - "entities": list(all_entities.values()), - "relationships": [{"source": s, "target": t, "label": l} for s, t, l in all_relationships] - } diff --git a/rag_system/indexing/latechunk.py b/rag_system/indexing/latechunk.py index 094a5243..054d56ed 100644 --- a/rag_system/indexing/latechunk.py +++ b/rag_system/indexing/latechunk.py @@ -21,21 +21,25 @@ class LateChunkEncoder: """Generate late-chunked embeddings given character-offset spans.""" - def __init__(self, model_name: str = "Qwen/Qwen3-Embedding-0.6B", *, max_tokens: int = 8192) -> None: + def __init__(self, model_name: str | None = None, *, max_tokens: int = 8192) -> None: + if not model_name: + from rag_system.main import EXTERNAL_MODELS + model_name = EXTERNAL_MODELS["embedding_model"] self.model_name = model_name self.max_len = max_tokens - self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") - # Back-compat: allow short alias without repo namespace - repo_id = model_name - if "/" not in model_name and not model_name.startswith("Qwen/"): - # map common alias to official repo - alias_map = { - "qwen3-embedding-0.6b": "Qwen/Qwen3-Embedding-0.6B", - } - repo_id = alias_map.get(model_name.lower(), model_name) - - self.tokenizer = AutoTokenizer.from_pretrained(repo_id, trust_remote_code=True) - self.model = AutoModel.from_pretrained(repo_id, trust_remote_code=True) + if torch.cuda.is_available(): + self.device = torch.device("cuda") + elif getattr(torch.backends, "mps", None) and torch.backends.mps.is_available(): + self.device = torch.device("mps") + else: + self.device = torch.device("cpu") + + self.tokenizer = AutoTokenizer.from_pretrained(model_name, trust_remote_code=True) + self.model = AutoModel.from_pretrained( + model_name, + trust_remote_code=True, + torch_dtype=torch.float16 if self.device.type != "cpu" else None, + ) self.model.to(self.device) self.model.eval() @@ -53,29 +57,51 @@ def encode(self, text: str, chunk_spans: List[Tuple[int, int]]) -> List[np.ndarr if not chunk_spans: return [] - # Tokenise and obtain per-token hidden states - inputs = self.tokenizer( - text, - return_tensors="pt", - return_offsets_mapping=True, - truncation=True, - max_length=self.max_len, - ) - inputs = {k: v.to(self.device) for k, v in inputs.items()} - offsets = inputs.pop("offset_mapping").squeeze(0).cpu().tolist() # (seq_len, 2) + # Tokenise the whole document once WITHOUT truncation, then run the + # model over overlapping windows of max_len tokens. A single truncated + # pass used to leave every span past max_len without tokens; the + # windows keep global character offsets so each span maps to real + # token vectors no matter how long the document is. + encoding = self.tokenizer(text, return_offsets_mapping=True, add_special_tokens=False) + input_ids = encoding["input_ids"] + offsets = encoding["offset_mapping"] # global (char_start, char_end) per token + total_tokens = len(input_ids) + + if total_tokens == 0: + # Empty/whitespace document: nothing to pool, return zero vectors. + hidden = int(getattr(getattr(self.model, "config", None), "hidden_size", 0) or 0) + return [np.zeros(hidden, dtype="float32") for _ in chunk_spans] + + overlap = self.max_len // 4 + stride = max(1, self.max_len - overlap) - out = self.model(**inputs) - last_hidden = out.last_hidden_state.squeeze(0) # (seq_len, dim) - last_hidden = last_hidden.cpu() + # Run each window and keep the per-token vectors (first window to + # cover a token wins; overlap tokens are recomputed but equivalent). + token_vectors = [None] * total_tokens + for win_start in range(0, total_tokens, stride): + win_end = min(win_start + self.max_len, total_tokens) + window_ids = torch.tensor([input_ids[win_start:win_end]], dtype=torch.long, device=self.device) + attention = torch.ones_like(window_ids) + out = self.model(input_ids=window_ids, attention_mask=attention) + last_hidden = out.last_hidden_state.squeeze(0) # (win_len, dim) + last_hidden = last_hidden.float().cpu() + for i in range(win_end - win_start): + if token_vectors[win_start + i] is None: + token_vectors[win_start + i] = last_hidden[i] + if win_end >= total_tokens: + break # For each chunk span, gather token indices belonging to it vectors: List[np.ndarray] = [] for start_char, end_char in chunk_spans: token_indices = [i for i, (s, e) in enumerate(offsets) if s >= start_char and e <= end_char] if not token_indices: - # Fallback: if tokenizer lost the span (e.g. due to trimming) just average CLS + SEP - token_indices = [0] - chunk_vec = last_hidden[token_indices].mean(dim=0).numpy().astype("float32") + # Degenerate span (e.g. whitespace the tokenizer folded into a + # neighbouring token): pool the nearest preceding token rather + # than the old always-token-0 fallback. + preceding = [i for i, (s, e) in enumerate(offsets) if e <= start_char] + token_indices = [preceding[-1]] if preceding else [0] + chunk_vec = torch.stack([token_vectors[i] for i in token_indices]).mean(dim=0).numpy().astype("float32") # Check for NaN or infinite values if np.isnan(chunk_vec).any() or np.isinf(chunk_vec).any(): diff --git a/rag_system/indexing/multimodal.py b/rag_system/indexing/multimodal.py deleted file mode 100644 index b2a89945..00000000 --- a/rag_system/indexing/multimodal.py +++ /dev/null @@ -1,124 +0,0 @@ -import fitz # PyMuPDF -from PIL import Image -import torch -import os -from typing import List, Dict, Any - -from rag_system.indexing.embedders import LanceDBManager, VectorIndexer -from rag_system.indexing.representations import QwenEmbedder - - -from transformers import ColPaliForRetrieval, ColPaliProcessor, Qwen2TokenizerFast - -class LocalVisionModel: - """ - A wrapper for a local vision model (ColPali) from the transformers library. - """ - def __init__(self, model_name: str = "vidore/colqwen2-v1.0", device: str = "cpu"): - print(f"Initializing local vision model '{model_name}' on device '{device}'.") - self.device = device - self.model = ColPaliForRetrieval.from_pretrained(model_name).to(self.device).eval() - self.tokenizer = Qwen2TokenizerFast.from_pretrained(model_name) - self.image_processor = ColPaliProcessor.from_pretrained(model_name).image_processor - self.processor = ColPaliProcessor(tokenizer=self.tokenizer, image_processor=self.image_processor) - print("Local vision model loaded successfully.") - - def embed_image(self, image: Image.Image) -> torch.Tensor: - """ - Generates a multi-vector embedding for a single image. - """ - inputs = self.processor(text="", images=image, return_tensors="pt").to(self.device) - with torch.no_grad(): - image_embeds = self.model.get_image_features(**inputs) - return image_embeds - - -class MultimodalProcessor: - """ - Processes PDFs into separate text and image embeddings using local models. - """ - def __init__(self, vision_model: LocalVisionModel, text_embedder: QwenEmbedder, db_manager: LanceDBManager): - self.vision_model = vision_model - self.text_embedder = text_embedder - self.text_vector_indexer = VectorIndexer(db_manager) - self.image_vector_indexer = VectorIndexer(db_manager) - - def process_and_index( - self, - pdf_path: str, - text_table_name: str, - image_table_name: str - ): - print(f"\n--- Processing PDF for multimodal indexing: {os.path.basename(pdf_path)} ---") - doc = fitz.open(pdf_path) - document_id = os.path.basename(pdf_path) - - all_pages_text_chunks = [] - all_pages_images = [] - - for page_num in range(len(doc)): - page = doc.load_page(page_num) - - # 1. Extract Text - text = page.get_text("text") - if not text.strip(): - text = f"Page {page_num + 1} contains no extractable text." - - all_pages_text_chunks.append({ - "chunk_id": f"{document_id}_page_{page_num+1}", - "text": text, - "metadata": {"document_id": document_id, "page_number": page_num + 1} - }) - - # 2. Extract Image - pix = page.get_pixmap() - img = Image.frombytes("RGB", [pix.width, pix.height], pix.samples) - all_pages_images.append(img) - - # --- Batch Indexing --- - # Index all text chunks - if all_pages_text_chunks: - text_embeddings = self.text_embedder.create_embeddings([c['text'] for c in all_pages_text_chunks]) - self.text_vector_indexer.index(text_table_name, all_pages_text_chunks, text_embeddings) - print(f"Indexed {len(all_pages_text_chunks)} text pages into '{text_table_name}'.") - - # Index all images - if all_pages_images: - image_embeddings = self.vision_model.create_image_embeddings(all_pages_images) - # We use the text chunks as placeholders for metadata - self.image_vector_indexer.index(image_table_name, all_pages_text_chunks, image_embeddings) - print(f"Indexed {len(all_pages_images)} image pages into '{image_table_name}'.") - -if __name__ == '__main__': - # This test requires an internet connection to download the models. - try: - # 1. Setup models and dependencies - text_embedder = QwenEmbedder() - vision_model = LocalVisionModel() - db_manager = LanceDBManager(db_path="./rag_system/index_store/lancedb") - - # 2. Create a dummy PDF - dummy_pdf_path = "multimodal_test.pdf" - doc = fitz.open() - page = doc.new_page() - page.insert_text((50, 72), "This is a test page with text and an image.") - doc.save(dummy_pdf_path) - - # 3. Run the processor - processor = MultimodalProcessor(vision_model, text_embedder, db_manager) - processor.process_and_index( - pdf_path=dummy_pdf_path, - text_table_name="test_text_pages", - image_table_name="test_image_pages" - ) - - # 4. Verify - print("\n--- Verification ---") - text_tbl = db_manager.get_table("test_text_pages") - img_tbl = db_manager.get_table("test_image_pages") - print(f"Text table has {len(text_tbl)} rows.") - print(f"Image table has {len(img_tbl)} rows.") - - except Exception as e: - print(f"\nAn error occurred during the multimodal test: {e}") - print("Please ensure you have an internet connection for model downloads.") \ No newline at end of file diff --git a/rag_system/indexing/overview_builder.py b/rag_system/indexing/overview_builder.py index 27ede9b3..26b74500 100644 --- a/rag_system/indexing/overview_builder.py +++ b/rag_system/indexing/overview_builder.py @@ -1,10 +1,107 @@ from __future__ import annotations import os, json, logging, re -from typing import List, Dict, Any +from typing import Any, Dict, List, Optional logger = logging.getLogger(__name__) +# --------------------------------------------------------------------------- +# Embedded-overview sidecar (roadmap item 4.3) +# --------------------------------------------------------------------------- +# The overviews themselves are one JSONL line per document, appended as each +# document is chunked. The overview *prefilter* needs them as vectors, so a +# sidecar `.npz` is written next to the JSONL at the end of an index build: +# +# index_store/overviews/<index_id>.jsonl the overviews +# index_store/overviews/<index_id>.vectors.npz doc_ids + L2-normalized vectors +# +# It is a sidecar and not a LanceDB table because it is one row per *document* +# (tens, not thousands), it is rebuilt wholesale rather than queried, and a +# missing sidecar has to be a graceful no-op rather than a schema problem. +# +# Vectors are written by the DOCUMENT-side embedder (no instruction prefix), +# matching how chunks are embedded, so a query-side vector can be compared +# against them with the same asymmetry the chunk index uses. + +VECTORS_SUFFIX = ".vectors.npz" + + +def overview_vectors_path(overview_path: str) -> str: + """The sidecar path for an overviews JSONL path.""" + if overview_path.endswith(".jsonl"): + return overview_path[: -len(".jsonl")] + VECTORS_SUFFIX + return overview_path + VECTORS_SUFFIX + + +def read_overviews(overview_path: str) -> Dict[str, str]: + """``{doc_id: overview}`` from an appended JSONL; the last line per doc wins.""" + out: Dict[str, str] = {} + try: + with open(overview_path, "r", encoding="utf-8") as fh: + for line in fh: + line = line.strip() + if not line: + continue + try: + record = json.loads(line) + except ValueError: + continue + doc_id = record.get("doc_id") + overview = (record.get("overview") or "").strip() + if not doc_id or not overview: + continue + if overview.startswith("Failed to generate overview"): + continue + out[doc_id] = overview + except OSError: + return {} + return out + + +def load_overview_vectors(vectors_path: str) -> Optional[Dict[str, Any]]: + """Read a sidecar. Returns ``None`` for any missing or unreadable file.""" + if not vectors_path or not os.path.exists(vectors_path): + return None + try: + import numpy as np + + with np.load(vectors_path, allow_pickle=False) as data: + doc_ids = [str(d) for d in data["doc_ids"].tolist()] + vectors = np.asarray(data["vectors"], dtype="float32") + meta_raw = data["meta"].item() if "meta" in data else "{}" + except Exception as e: # unreadable sidecar must never break retrieval + logger.warning("Could not read overview vectors %s: %s", vectors_path, e) + return None + if len(doc_ids) != len(vectors) or not doc_ids: + return None + try: + meta = json.loads(meta_raw.decode("utf-8") if isinstance(meta_raw, bytes) else str(meta_raw)) + except ValueError: + meta = {} + return {"doc_ids": doc_ids, "vectors": vectors, "meta": meta, "path": vectors_path} + + +def write_overview_vectors(vectors_path: str, doc_ids: List[str], vectors, + embedding_model: Optional[str]) -> None: + import numpy as np + + out_dir = os.path.dirname(vectors_path) + if out_dir: + os.makedirs(out_dir, exist_ok=True) + meta = json.dumps({"embedding_model": embedding_model, "normalized": True}) + # Write to a temp file in the same directory, then os.replace: a crash + # mid-savez must not leave a truncated sidecar behind. + tmp_path = vectors_path + ".tmp" + with open(tmp_path, "wb") as fh: + np.savez( + fh, + doc_ids=np.array(doc_ids, dtype=object).astype("U"), + vectors=np.asarray(vectors, dtype="float32"), + meta=np.array(meta), + ) + os.replace(tmp_path, vectors_path) + + class OverviewBuilder: """Generates and stores a one-paragraph overview for each document. The overview is derived from the first *n* chunks of the document. @@ -18,7 +115,7 @@ class OverviewBuilder: "DOCUMENT_START:\n{text}\n\nOVERVIEW:" ) - def __init__(self, llm_client, model: str = "qwen3:0.6b", first_n_chunks: int = 5, + def __init__(self, llm_client, model: str, first_n_chunks: int = 5, out_path: str | None = None): if out_path is None: out_path = "index_store/overviews/overviews.jsonl" @@ -26,7 +123,9 @@ def __init__(self, llm_client, model: str = "qwen3:0.6b", first_n_chunks: int = self.model = model self.first_n = first_n_chunks self.out_path = out_path - os.makedirs(os.path.dirname(out_path), exist_ok=True) + out_dir = os.path.dirname(out_path) + if out_dir: + os.makedirs(out_dir, exist_ok=True) def build_and_store(self, doc_id: str, chunks: List[Dict[str, Any]]): if not chunks: @@ -41,7 +140,57 @@ def build_and_store(self, doc_id: str, chunks: List[Dict[str, Any]]): except Exception as e: summary = f"Failed to generate overview: {e}" record = {"doc_id": doc_id, "overview": summary.strip()} - with open(self.out_path, "a", encoding="utf-8") as f: - f.write(json.dumps(record, ensure_ascii=False) + "\n") + # Atomic append: rewrite the JSONL through a temp file + os.replace so + # a crash mid-write can never leave a truncated file. One line per + # document is preserved, so the last-line-wins read semantics are + # unchanged. + tmp_path = f"{self.out_path}.{os.getpid()}.tmp" + with open(tmp_path, "w", encoding="utf-8") as tmp: + try: + with open(self.out_path, "r", encoding="utf-8") as src: + for existing in src: + tmp.write(existing) + except OSError: + pass # first write: no existing JSONL yet + tmp.write(json.dumps(record, ensure_ascii=False) + "\n") + os.replace(tmp_path, self.out_path) + + logger.info(f"๐Ÿ“„ Overview generated for {doc_id} (stored in {self.out_path})") + + # ------------------------------------------------------------------ + # Embedded-overview sidecar (roadmap item 4.3) + # ------------------------------------------------------------------ + + @property + def vectors_path(self) -> str: + return overview_vectors_path(self.out_path) + + def embed_and_store_vectors(self, embedder, embedding_model: Optional[str] = None) -> int: + """Embed every overview in the JSONL and (re)write the ``.npz`` sidecar. + + Rebuilt wholesale rather than appended: the JSONL is append-only, so a + re-indexed document has several lines and only the last one is current. + Returns the number of documents embedded (0 when there is nothing to do). + """ + overviews = read_overviews(self.out_path) + if not overviews: + logger.info("No usable overviews in %s; skipping the vector sidecar.", self.out_path) + return 0 + + import numpy as np + + doc_ids = sorted(overviews) + texts = [overviews[d] for d in doc_ids] + vectors = np.asarray(embedder.create_embeddings(texts), dtype="float32") + if vectors.ndim != 2 or len(vectors) != len(doc_ids): + logger.warning("Overview embedding returned an unexpected shape; sidecar not written.") + return 0 + + norms = np.linalg.norm(vectors, axis=1, keepdims=True) + norms[norms == 0.0] = 1.0 + vectors = vectors / norms - logger.info(f"๐Ÿ“„ Overview generated for {doc_id} (stored in {self.out_path})") \ No newline at end of file + model_name = embedding_model or getattr(embedder, "model_name", None) + write_overview_vectors(self.vectors_path, doc_ids, vectors, model_name) + logger.info("๐Ÿงญ Embedded %d document overview(s) into %s", len(doc_ids), self.vectors_path) + return len(doc_ids) diff --git a/rag_system/indexing/representations.py b/rag_system/indexing/representations.py index a3e5ce38..36ed9d71 100644 --- a/rag_system/indexing/representations.py +++ b/rag_system/indexing/representations.py @@ -11,13 +11,72 @@ def create_embeddings(self, texts: List[str]) -> np.ndarray: ... # Global cache for models - use dict to cache by model name _MODEL_CACHE = {} +# --------------------------------------------------------------------------- +# Query-side instruction prefix +# --------------------------------------------------------------------------- +# Instruction-tuned decoder embedders (Qwen3-Embedding, microsoft/harrier-oss-v1) +# are trained with an instruction on the QUERY side only. Both model cards use +# the identical wire format and the identical MS-MARCO-style retrieval task +# string, and both state explicitly that documents must be embedded WITHOUT any +# instruction. +# +# query -> "Instruct: {task}\nQuery: {text}" +# document -> "{text}" (unchanged, always) +# +# The asymmetry is what keeps this change index-compatible: an index built +# before the prefix existed stays valid, because nothing on the document side +# moves. Only the query vector changes. +# +# An embedder instance carries the instruction; it does not decide per call. +# The indexing pipeline builds its embedder without one, the retrieval pipeline +# builds its embedder with one, and neither can leak into the other. + +QUERY_PROMPT_TEMPLATE = "Instruct: {instruction}\nQuery: {text}" + +# The official retrieval task description used by both model families. +DEFAULT_RETRIEVAL_INSTRUCTION = ( + "Given a web search query, retrieve relevant passages that answer the query" +) + +# Model-name fragments whose families are instruction-tuned in this format. +_INSTRUCTION_TUNED_FAMILIES = ("qwen3-embedding", "harrier") + + +def default_query_instruction(model_name: str) -> str: + """The retrieval instruction a model family expects, or "" when it wants none. + + Returning "" (not None) is deliberate: "" means "this model takes no + instruction", which is a decision, whereas None means "nobody decided yet" + and is what callers pass to ask for this default. + """ + name = (model_name or "").lower() + if any(fragment in name for fragment in _INSTRUCTION_TUNED_FAMILIES): + return DEFAULT_RETRIEVAL_INSTRUCTION + return "" + + +def apply_query_instruction(texts: List[str], instruction: str | None) -> List[str]: + """Prefix every text with the instruction block, or return them untouched.""" + if not instruction: + return texts + return [QUERY_PROMPT_TEMPLATE.format(instruction=instruction, text=t) for t in texts] + # --- New Ollama Embedder --- class QwenEmbedder(EmbeddingModel): """ An embedding model that uses a local Hugging Face transformer model. """ - def __init__(self, model_name: str = "Qwen/Qwen3-Embedding-0.6B"): + MAX_TOKENS = 8192 + + def __init__(self, model_name: str | None = None, query_instruction: str | None = None): + if not model_name: + from rag_system.main import EXTERNAL_MODELS + model_name = EXTERNAL_MODELS["embedding_model"] self.model_name = model_name + # "" / None => this instance embeds raw text (the document side). + # A non-empty string => this instance is a QUERY embedder and prefixes + # every text it is given with the instruction block. + self.query_instruction = query_instruction or "" # Auto-select the best available device: CUDA > MPS > CPU if torch.cuda.is_available(): self.device = "cuda" @@ -39,36 +98,51 @@ def __init__(self, model_name: str = "Qwen/Qwen3-Embedding-0.6B"): print(f"QwenEmbedder weights loaded and cached for {model_name}.") else: print(f"Reusing cached QwenEmbedder weights for {model_name}.") - + self.tokenizer, self.model = _MODEL_CACHE[model_name] + # Some tokenizers report a sentinel model_max_length; clamp it so that + # truncation=True actually truncates. + reported = getattr(self.tokenizer, "model_max_length", None) + self.max_length = min(reported, self.MAX_TOKENS) if isinstance(reported, int) and reported > 0 else self.MAX_TOKENS def create_embeddings(self, texts: List[str]) -> np.ndarray: print(f"Generating {len(texts)} embeddings with {self.model_name} model...") - inputs = self.tokenizer(texts, padding=True, truncation=True, return_tensors="pt").to(self.device) + texts = apply_query_instruction(texts, self.query_instruction) + inputs = self.tokenizer( + texts, + padding=True, + truncation=True, + max_length=self.max_length, + return_tensors="pt", + ).to(self.device) with torch.no_grad(): outputs = self.model(**inputs) last_hidden = outputs.last_hidden_state # [B, seq, dim] - # Pool via last valid token per sequence (recommended for Qwen3) - seq_len = inputs["attention_mask"].sum(dim=1) - 1 # index of last token - batch_indices = torch.arange(last_hidden.size(0), device=self.device) - embeddings = last_hidden[batch_indices, seq_len] - + # Last-token pooling (recommended for Qwen3-Embedding). The tokenizer + # pads on the left, in which case the final column is the last real + # token for every row; handle right padding too for safety. + attention_mask = inputs["attention_mask"] + left_padded = bool(attention_mask[:, -1].min().item()) + if left_padded: + embeddings = last_hidden[:, -1] + else: + seq_len = attention_mask.sum(dim=1) - 1 # index of last token + batch_indices = torch.arange(last_hidden.size(0), device=last_hidden.device) + embeddings = last_hidden[batch_indices, seq_len] + # Convert to numpy and validate - embeddings_np = embeddings.cpu().numpy() - - # Check for NaN or infinite values + embeddings_np = embeddings.float().cpu().numpy() + + # Warn only โ€” do NOT zero-fill. An all-zero vector would pass the + # NaN/Inf filter in VectorIndexer.index() and be indexed as a real + # row; leaving the invalid values in place lets that filter skip the + # affected chunks instead. if np.isnan(embeddings_np).any(): - print(f"โš ๏ธ Warning: NaN values detected in embeddings from {self.model_name}") - # Replace NaN values with zeros - embeddings_np = np.nan_to_num(embeddings_np, nan=0.0, posinf=0.0, neginf=0.0) - print(f"๐Ÿ”„ Replaced NaN values with zeros") - + print(f"โš ๏ธ Warning: NaN values detected in embeddings from {self.model_name}; affected chunks will be skipped at indexing time") + if np.isinf(embeddings_np).any(): - print(f"โš ๏ธ Warning: Infinite values detected in embeddings from {self.model_name}") - # Replace infinite values with zeros - embeddings_np = np.nan_to_num(embeddings_np, nan=0.0, posinf=0.0, neginf=0.0) - print(f"๐Ÿ”„ Replaced infinite values with zeros") - + print(f"โš ๏ธ Warning: Infinite values detected in embeddings from {self.model_name}; affected chunks will be skipped at indexing time") + return embeddings_np class EmbeddingGenerator: @@ -105,10 +179,13 @@ def process_text_batch(text_batch): class OllamaEmbedder(EmbeddingModel): """Call Ollama's /api/embeddings endpoint for each text.""" - def __init__(self, model_name: str, host: str | None = None, timeout: int = 60): + def __init__(self, model_name: str, host: str | None = None, timeout: int = 60, + query_instruction: str | None = None): self.model_name = model_name self.host = (host or os.getenv("OLLAMA_HOST") or "http://localhost:11434").rstrip("/") self.timeout = timeout + # Same contract as QwenEmbedder: set on the query-side instance only. + self.query_instruction = query_instruction or "" def _embed_single(self, text: str): import requests, numpy as np, json @@ -124,31 +201,37 @@ def _embed_single(self, text: str): def create_embeddings(self, texts: List[str]): import numpy as np + texts = apply_query_instruction(texts, self.query_instruction) vectors = [self._embed_single(t) for t in texts] embeddings_np = np.vstack(vectors) - - # Check for NaN or infinite values + + # Warn only โ€” do NOT zero-fill. An all-zero vector would pass the + # NaN/Inf filter in VectorIndexer.index() and be indexed as a real + # row; leaving the invalid values in place lets that filter skip the + # affected chunks instead. if np.isnan(embeddings_np).any(): - print(f"โš ๏ธ Warning: NaN values detected in Ollama embeddings from {self.model_name}") - # Replace NaN values with zeros - embeddings_np = np.nan_to_num(embeddings_np, nan=0.0, posinf=0.0, neginf=0.0) - print(f"๐Ÿ”„ Replaced NaN values with zeros") - + print(f"โš ๏ธ Warning: NaN values detected in Ollama embeddings from {self.model_name}; affected chunks will be skipped at indexing time") + if np.isinf(embeddings_np).any(): - print(f"โš ๏ธ Warning: Infinite values detected in Ollama embeddings from {self.model_name}") - # Replace infinite values with zeros - embeddings_np = np.nan_to_num(embeddings_np, nan=0.0, posinf=0.0, neginf=0.0) - print(f"๐Ÿ”„ Replaced infinite values with zeros") - + print(f"โš ๏ธ Warning: Infinite values detected in Ollama embeddings from {self.model_name}; affected chunks will be skipped at indexing time") + return embeddings_np -def select_embedder(model_name: str, ollama_host: str | None = None): - """Return appropriate EmbeddingModel implementation for the given name.""" +def select_embedder(model_name: str, ollama_host: str | None = None, + query_instruction: str | None = None): + """Return appropriate EmbeddingModel implementation for the given name. + + ``query_instruction`` is the query-side instruction prefix. Leave it unset + (the default) for document-side embedders โ€” that is what keeps indexes + stable across this change. Callers on the query path pass the instruction + explicitly; see ``RetrievalPipeline._get_text_embedder``. + """ if "/" in model_name or model_name.startswith("http"): # Treat as HF model path - return QwenEmbedder(model_name=model_name) + return QwenEmbedder(model_name=model_name, query_instruction=query_instruction) # Otherwise assume it's an Ollama tag - return OllamaEmbedder(model_name=model_name, host=ollama_host) + return OllamaEmbedder(model_name=model_name, host=ollama_host, + query_instruction=query_instruction) if __name__ == '__main__': print("representations.py cleaned up.") diff --git a/rag_system/ingestion/chunking.py b/rag_system/ingestion/chunking.py index 6e55ff0a..dbf01e6a 100644 --- a/rag_system/ingestion/chunking.py +++ b/rag_system/ingestion/chunking.py @@ -8,21 +8,19 @@ class MarkdownRecursiveChunker: and embeds document-level metadata into each chunk. """ - def __init__(self, max_chunk_size: int = 1500, min_chunk_size: int = 200, tokenizer_model: str = "Qwen/Qwen3-Embedding-0.6B"): + def __init__(self, max_chunk_size: int = 1500, min_chunk_size: int = 200, tokenizer_model: str | None = None): self.max_chunk_size = max_chunk_size self.min_chunk_size = min_chunk_size self.split_priority = ["\n## ", "\n### ", "\n#### ", "```", "\n\n"] - - repo_id = tokenizer_model - if "/" not in tokenizer_model and not tokenizer_model.startswith("Qwen/"): - repo_id = { - "qwen3-embedding-0.6b": "Qwen/Qwen3-Embedding-0.6B", - }.get(tokenizer_model.lower(), tokenizer_model) - + + if not tokenizer_model: + from rag_system.main import EXTERNAL_MODELS + tokenizer_model = EXTERNAL_MODELS["embedding_model"] + try: - self.tokenizer = AutoTokenizer.from_pretrained(repo_id, trust_remote_code=True) + self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_model, trust_remote_code=True) except Exception as e: - print(f"Warning: Failed to load tokenizer {repo_id}: {e}") + print(f"Warning: Failed to load tokenizer {tokenizer_model}: {e}") print("Falling back to character-based approximation (4 chars โ‰ˆ 1 token)") self.tokenizer = None @@ -41,17 +39,21 @@ def _split_text(self, text: str, separators: List[str]) -> List[str]: new_chunks = [] for chunk in chunks_to_process: if self._token_len(chunk) > self.max_chunk_size: - sub_chunks = re.split(f'({sep})', chunk) + # re.split with a capture group interleaves segments and + # separators: [seg0, sep, seg1, sep, seg2, ...]. Reattach + # each separator to the segment that FOLLOWS it, keeping + # every segment. (The previous loop advanced 3 positions + # after consuming 2, silently dropping seg0 and every + # other body segment of any document large enough to + # split โ€” real corpora lost ~half their text.) + sub_chunks = re.split(f'({re.escape(sep)})', chunk) combined = [] - i = 0 - while i < len(sub_chunks): - if i + 1 < len(sub_chunks) and sub_chunks[i+1] == sep: - combined.append(sub_chunks[i+1] + sub_chunks[i+2]) - i += 3 - else: - if sub_chunks[i]: - combined.append(sub_chunks[i]) - i += 1 + if sub_chunks and sub_chunks[0]: + combined.append(sub_chunks[0]) + for j in range(1, len(sub_chunks) - 1, 2): + piece = sub_chunks[j] + sub_chunks[j + 1] + if piece: + combined.append(piece) new_chunks.extend(combined) else: new_chunks.append(chunk) @@ -100,9 +102,10 @@ def chunk(self, text: str, document_id: str, document_metadata: Optional[Dict[st test_chunk = current_chunk + chunk_text if current_chunk else chunk_text if not current_chunk or self._token_len(test_chunk) <= self.max_chunk_size: current_chunk = test_chunk - elif self._token_len(current_chunk) < self.min_chunk_size: - current_chunk = test_chunk else: + # An undersized current_chunk is emitted as-is when merging it + # forward would push the result past max_chunk_size โ€” a merge + # must never overshoot the token budget. merged_chunks_text.append(current_chunk) current_chunk = chunk_text if current_chunk: @@ -128,8 +131,20 @@ def chunk(self, text: str, document_id: str, document_metadata: Optional[Dict[st def create_contextual_window(all_chunks: List[Dict[str, Any]], chunk_index: int, window_size: int = 1) -> str: if not (0 <= chunk_index < len(all_chunks)): raise ValueError("chunk_index is out of bounds.") - start = max(0, chunk_index - window_size) - end = min(len(all_chunks), chunk_index + window_size + 1) + + def _doc_id(chunk: Dict[str, Any]): + metadata = chunk.get("metadata") or {} + return metadata.get("document_id", chunk.get("document_id")) + + # The flat list spans many documents; clamp the window to chunks of the + # same document so one document's context never leaks into another's. + doc_id = _doc_id(all_chunks[chunk_index]) + start = chunk_index + while start > 0 and chunk_index - (start - 1) <= window_size and _doc_id(all_chunks[start - 1]) == doc_id: + start -= 1 + end = chunk_index + 1 + while end < len(all_chunks) and end - chunk_index <= window_size and _doc_id(all_chunks[end]) == doc_id: + end += 1 context_chunks = all_chunks[start:end] return " ".join([chunk['text'] for chunk in context_chunks]) diff --git a/rag_system/ingestion/docling_chunker.py b/rag_system/ingestion/docling_chunker.py index 4a27ff44..08f6e57b 100644 --- a/rag_system/ingestion/docling_chunker.py +++ b/rag_system/ingestion/docling_chunker.py @@ -1,40 +1,50 @@ from __future__ import annotations -"""Docling-aware chunker (simplified). +"""Docling-aware chunker. -For now we proxy the old MarkdownRecursiveChunker but add: -โ€ข sentence-aware packing to max_tokens with overlap -โ€ข breadcrumb metadata stubs so downstream code already handles them +Two entry points: +โ€ข chunk_document(doc) walks a DoclingDocument element tree, emitting tables and + code as atomic chunks and token-packing paragraphs up to max_tokens. +โ€ข chunk()/split_markdown() fall back to MarkdownRecursiveChunker plus + sentence-aware packing when only Markdown is available. -In a follow-up we can replace the internals with true Docling element-tree -walking once the PDFConverter returns structured nodes. +Both attach heading-path / block-type metadata to every chunk. """ -from typing import List, Dict, Any, Tuple -import math +from typing import List, Dict, Any import re -from itertools import islice from rag_system.ingestion.chunking import MarkdownRecursiveChunker from transformers import AutoTokenizer class DoclingChunker: - def __init__(self, *, max_tokens: int = 512, overlap: int = 1, tokenizer_model: str = "Qwen/Qwen3-Embedding-0.6B"): + def __init__(self, *, max_tokens: int = 512, overlap: int = 1, tokenizer_model: str | None = None): self.max_tokens = max_tokens self.overlap = overlap # sentences of overlap - repo_id = tokenizer_model - if "/" not in tokenizer_model and not tokenizer_model.startswith("Qwen/"): - repo_id = { - "qwen3-embedding-0.6b": "Qwen/Qwen3-Embedding-0.6B", - }.get(tokenizer_model.lower(), tokenizer_model) - + + if not tokenizer_model: + from rag_system.main import EXTERNAL_MODELS + tokenizer_model = EXTERNAL_MODELS["embedding_model"] + try: - self.tokenizer = AutoTokenizer.from_pretrained(repo_id, trust_remote_code=True) + self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_model, trust_remote_code=True) except Exception as e: - print(f"Warning: Failed to load tokenizer {repo_id}: {e}") + print(f"Warning: Failed to load tokenizer {tokenizer_model}: {e}") print("Falling back to character-based approximation (4 chars โ‰ˆ 1 token)") self.tokenizer = None # Fallback simple sentence splitter (period, question, exclamation, newline) self._sent_re = re.compile(r"(?<=[\.\!\?])\s+|\n+") - self.legacy = MarkdownRecursiveChunker(max_chunk_size=10_000, min_chunk_size=100) + # Markdown fallback chunker, built lazily on first use so the common + # Docling path doesn't pay for a second tokenizer load. + self._legacy: MarkdownRecursiveChunker | None = None + self._legacy_tokenizer_model = tokenizer_model + + @property + def legacy(self) -> MarkdownRecursiveChunker: + if self._legacy is None: + self._legacy = MarkdownRecursiveChunker( + max_chunk_size=10_000, min_chunk_size=100, + tokenizer_model=self._legacy_tokenizer_model, + ) + return self._legacy # ------------------------------------------------------------------ def _token_len(self, text: str) -> int: @@ -53,14 +63,19 @@ def split_markdown(self, markdown: str, *, document_id: str, metadata: Dict[str, sentences = [s.strip() for s in self._sent_re.split(ch["text"]) if s.strip()] if not sentences: continue - window: List[str] = [] - while sentences: - # Add until over limit - while sentences and self._token_len(" ".join(window + [sentences[0]])) <= self.max_tokens: - window.append(sentences.pop(0)) - if not window: # single sentence > limit โ†’ hard cut - window.append(sentences.pop(0)) - chunk_text = " ".join(window) + # Index-based window over the sentence list. Each iteration emits + # sentences[i:j] and the next window starts strictly after i, so + # the loop always makes progress and cannot re-emit the same + # window forever (the old queue-prepend version could). + i = 0 + while i < len(sentences): + # Grow the window [i..j) until the next sentence would exceed the limit + j = i + while j < len(sentences) and self._token_len(" ".join(sentences[i:j + 1])) <= self.max_tokens: + j += 1 + if j == i: # single sentence > limit โ†’ hard cut + j = i + 1 + chunk_text = " ".join(sentences[i:j]) new_chunk = { "chunk_id": f"{document_id}_{global_idx}", "text": chunk_text, @@ -75,11 +90,11 @@ def split_markdown(self, markdown: str, *, document_id: str, metadata: Dict[str, } new_chunks.append(new_chunk) global_idx += 1 - # Overlap: prepend last `overlap` sentences of the current window to the remaining queue - if self.overlap and sentences: - back = window[-self.overlap:] if self.overlap <= len(window) else window[:] - sentences = back + sentences - window = [] + if j >= len(sentences): + break + # Overlap: restart up to `overlap` sentences back, but never at + # or before i โ€” the window must strictly advance. + i = max(j - self.overlap, i + 1) if self.overlap else j return new_chunks # ------------------------------------------------------------------ @@ -88,8 +103,10 @@ def split_markdown(self, markdown: str, *, document_id: str, metadata: Dict[str, def chunk_document(self, doc, *, document_id: str, metadata: Dict[str, Any] | None = None) -> List[Dict[str, Any]]: """Walk a DoclingDocument and emit chunks. - Tables / Code / Figures are emitted as atomic chunks. - Paragraph-like nodes are sentence-packed to <= max_tokens. + Tables are emitted as atomic chunks, inline in reading order. + Section headers update the heading path and are not emitted. + Text-less items (pictures, groups) are skipped; everything with + text is token-packed up to max_tokens. """ metadata = metadata or {} @@ -125,9 +142,11 @@ def _add_chunk(text: str, block_type: str, heading_path: List[str], page_no: int }) global_idx += 1 - # The Docling API exposes .body which is a tree of nodes; we fall back to .texts/.tables lists if available + # Walk the document with docling's iterate_items(), which yields + # (item, level) in true reading order โ€” tables included inline. + # Attributes are read through getattr so unknown item types degrade + # gracefully; anything unexpected falls back to the markdown splitter. try: - # We walk doc.texts (reading order). We'll buffer consecutive paragraph items current_heading_path: List[str] = [] buffer: List[str] = [] buffer_tokens = 0 @@ -139,38 +158,42 @@ def flush_buffer(): _add_chunk(" ".join(buffer), "paragraph", heading_path=current_heading_path[:], page_no=buffer_page) buffer, buffer_tokens, buffer_page = [], 0, None - # Create quick lookup for table items by id to preserve later insertion order if needed - tables_by_anchor = { - getattr(t, "anchor_text_id", None): t - for t in getattr(doc, "tables", []) - if getattr(t, "anchor_text_id", None) is not None - } + def _page_of(item) -> int | None: + prov = getattr(item, "prov", None) or [] + return getattr(prov[0], "page_no", None) if prov else None + + def _emit_table(tbl) -> None: + try: + tbl_md = tbl.export_to_markdown(doc) # pass doc for deprecation compliance + except Exception: + tbl_md = tbl.export_to_markdown() if hasattr(tbl, "export_to_markdown") else str(tbl) + _add_chunk(tbl_md, "table", heading_path=current_heading_path[:], page_no=_page_of(tbl)) - for txt_item in getattr(doc, "texts", []): - # If this text item is a placeholder for a table anchor, emit table first - anchor_id = getattr(txt_item, "id", None) - if anchor_id in tables_by_anchor: + for item, _level in doc.iterate_items(): + label = getattr(item, "label", None) + label_value = getattr(label, "value", label) + + # Tables are atomic chunks, emitted where they appear in the flow + if label_value == "table": flush_buffer() - tbl = tables_by_anchor[anchor_id] - try: - tbl_md = tbl.export_to_markdown(doc) # pass doc for deprecation compliance - except Exception: - tbl_md = tbl.export_to_markdown() if hasattr(tbl, "export_to_markdown") else str(tbl) - _add_chunk(tbl_md, "table", heading_path=current_heading_path[:], page_no=getattr(tbl, "page_no", None)) - - role = getattr(txt_item, "role", None) - if role == "heading": + _emit_table(item) + continue + + # Section headings update the heading path; they are not content + if label_value == "section_header": flush_buffer() - level = getattr(txt_item, "level", 1) + level = getattr(item, "level", 1) or 1 current_heading_path = current_heading_path[: max(0, level - 1)] - current_heading_path.append(txt_item.text.strip()) + current_heading_path.append((getattr(item, "text", "") or "").strip()) continue # skip heading as content - text_piece = txt_item.text if hasattr(txt_item, "text") else str(txt_item) + text_piece = getattr(item, "text", None) + if not text_piece: + continue # pictures, groups and other text-less items piece_tokens = _token_len(text_piece) if piece_tokens > self.max_tokens: # very long paragraph flush_buffer() - _add_chunk(text_piece, "paragraph", heading_path=current_heading_path[:], page_no=getattr(txt_item, "page_no", None)) + _add_chunk(text_piece, "paragraph", heading_path=current_heading_path[:], page_no=_page_of(item)) continue if buffer_tokens + piece_tokens > self.max_tokens: @@ -179,19 +202,9 @@ def flush_buffer(): buffer.append(text_piece) buffer_tokens += piece_tokens if buffer_page is None: - buffer_page = getattr(txt_item, "page_no", None) + buffer_page = _page_of(item) flush_buffer() - - # Emit any remaining tables that were not anchored - for tbl in getattr(doc, "tables", []): - if tbl in tables_by_anchor.values(): - continue # already emitted - try: - tbl_md = tbl.export_to_markdown(doc) - except Exception: - tbl_md = tbl.export_to_markdown() if hasattr(tbl, "export_to_markdown") else str(tbl) - _add_chunk(tbl_md, "table", heading_path=current_heading_path[:], page_no=getattr(tbl, "page_no", None)) except Exception as e: print(f"โš ๏ธ Docling tree walk failed: {e}. Falling back to markdown splitter.") return self.split_markdown(doc.export_to_markdown(), document_id=document_id, metadata=metadata) diff --git a/rag_system/ingestion/document_converter.py b/rag_system/ingestion/document_converter.py index 78b2a9ce..2086b1dc 100644 --- a/rag_system/ingestion/document_converter.py +++ b/rag_system/ingestion/document_converter.py @@ -1,9 +1,72 @@ -from typing import List, Tuple, Dict, Any +from typing import List, Tuple, Dict, Any, Union +import os +import platform + +# Conversion result rows: the docling paths return +# (markdown, metadata, docling_document) triples; the plain-text path returns +# (markdown, metadata) pairs. Callers must handle both lengths. +ConversionResults = List[Union[Tuple[str, Dict[str, Any]], Tuple[str, Dict[str, Any], Any]]] + +# torch.compile's inductor backend has no MPS support; docling's layout model +# calls it and crashes on Apple Silicon unless dynamo is disabled up front. +if platform.system() == "Darwin": + os.environ.setdefault("TORCHDYNAMO_DISABLE", "1") + from docling.document_converter import DocumentConverter as DoclingConverter, PdfFormatOption -from docling.datamodel.pipeline_options import PdfPipelineOptions, OcrMacOptions +from docling.datamodel import pipeline_options as docling_options +from docling.datamodel.pipeline_options import PdfPipelineOptions from docling.datamodel.base_models import InputFormat import fitz # PyMuPDF for quick text inspection -import os +import importlib.util +import shutil + +# docling options class -> the module or binary its backend needs at runtime. +OCR_BACKENDS = [ + ("OcrMacOptions", "module", ("ocrmac",)), + ("EasyOcrOptions", "module", ("easyocr",)), + # rapidocr renamed its package: >=3.x installs as `rapidocr`, older + # releases as `rapidocr_onnxruntime` โ€” accept either. + ("RapidOcrOptions", "module", ("rapidocr", "rapidocr_onnxruntime")), + ("TesseractOcrOptions", "module", ("tesserocr",)), + ("TesseractCliOcrOptions", "binary", ("tesseract",)), +] + + +def build_ocr_options(): + """Pick an OCR engine docling can actually run on this host. + + OcrMac is only tried on macOS; the remaining engines are tried in order and + only if their backend is installed. Returns None when nothing is available, + in which case docling's own default OCR settings are used. + """ + for name, kind, dependencies in OCR_BACKENDS: + if name == "OcrMacOptions" and platform.system() != "Darwin": + continue + options_cls = getattr(docling_options, name, None) + if options_cls is None: + continue + if kind == "module" and not any(importlib.util.find_spec(d) for d in dependencies): + continue + if kind == "binary" and not any(shutil.which(d) for d in dependencies): + continue + try: + kwargs = {"force_full_page_ocr": True} + # docling's RapidOCR default is lang=['chinese']; pin an explicit + # recognition language (OCR_LANG env, comma-separated, to override). + if name == "RapidOcrOptions": + kwargs["lang"] = [ + l.strip() for l in os.getenv("OCR_LANG", "english").split(",") if l.strip() + ] + options = options_cls(**kwargs) + except Exception as e: + print(f"OCR engine {name} is not usable here: {e}") + continue + print(f"OCR engine: {name}") + return options + + print("No OCR engine available; using docling's default OCR settings.") + return None + class DocumentConverter: """ @@ -22,79 +85,96 @@ class DocumentConverter: } def __init__(self): - """Initializes the docling document converter with forced OCR enabled for macOS.""" + """Initializes one docling converter per path (no-OCR, OCR, general). + + Each converter is built independently so that a failure in one (typically + the OCR engine) does not disable the others. + """ + self.converter_no_ocr = self._build_pdf_converter(do_ocr=False) + self.converter_ocr = self._build_pdf_converter(do_ocr=True) try: - # --- Converter WITHOUT OCR (fast path) --- - pipeline_no_ocr = PdfPipelineOptions() - pipeline_no_ocr.do_ocr = False - format_no_ocr = { - InputFormat.PDF: PdfFormatOption(pipeline_options=pipeline_no_ocr) - } - self.converter_no_ocr = DoclingConverter(format_options=format_no_ocr) - - # --- Converter WITH OCR (fallback) --- - pipeline_ocr = PdfPipelineOptions() - pipeline_ocr.do_ocr = True - ocr_options = OcrMacOptions(force_full_page_ocr=True) - pipeline_ocr.ocr_options = ocr_options - format_ocr = { - InputFormat.PDF: PdfFormatOption(pipeline_options=pipeline_ocr) - } - self.converter_ocr = DoclingConverter(format_options=format_ocr) - self.converter_general = DoclingConverter() - - print("docling DocumentConverter(s) initialized (OCR + no-OCR + general).") except Exception as e: - print(f"Error initializing docling DocumentConverter(s): {e}") - self.converter_no_ocr = None - self.converter_ocr = None + print(f"Error initializing general docling converter: {e}") self.converter_general = None - def convert_to_markdown(self, file_path: str) -> List[Tuple[str, Dict[str, Any]]]: + available = [ + name for name, conv in ( + ("no-OCR", self.converter_no_ocr), + ("OCR", self.converter_ocr), + ("general", self.converter_general), + ) if conv is not None + ] + print(f"docling DocumentConverter(s) initialized ({', '.join(available) or 'none'}).") + + @staticmethod + def _build_pdf_converter(*, do_ocr: bool): + try: + pipeline = PdfPipelineOptions() + pipeline.do_ocr = do_ocr + if do_ocr: + ocr_options = build_ocr_options() + if ocr_options is not None: + pipeline.ocr_options = ocr_options + return DoclingConverter( + format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=pipeline)} + ) + except Exception as e: + print(f"Error initializing docling PDF converter (ocr={do_ocr}): {e}") + return None + + def convert_to_markdown(self, file_path: str) -> ConversionResults: """ Converts a document to a single Markdown string, preserving layout and tables. Supports PDF, DOCX, HTML, and other formats. """ - if not (self.converter_no_ocr and self.converter_ocr and self.converter_general): - print("docling converters not available. Skipping conversion.") - return [] - file_ext = os.path.splitext(file_path)[1].lower() if file_ext not in self.SUPPORTED_FORMATS: print(f"Unsupported file format: {file_ext}") return [] - + input_format = self.SUPPORTED_FORMATS[file_ext] - + if input_format == InputFormat.PDF: return self._convert_pdf_to_markdown(file_path) elif input_format == 'TXT': return self._convert_txt_to_markdown(file_path) else: return self._convert_general_to_markdown(file_path, input_format) - - def _convert_pdf_to_markdown(self, pdf_path: str) -> List[Tuple[str, Dict[str, Any]]]: + + def _convert_pdf_to_markdown(self, pdf_path: str) -> ConversionResults: """Convert PDF with OCR detection logic.""" - # Quick heuristic: if the PDF already contains a text layer, skip OCR for speed + # Quick heuristic: skip OCR for speed when the PDF mostly has a text + # layer. A *majority* of pages must carry text โ€” accepting any single + # text page let mixed text/scanned PDFs skip OCR and silently lose + # their scanned pages. def _pdf_has_text(path: str) -> bool: try: doc = fitz.open(path) - for page in doc: - if page.get_text("text").strip(): - return True + total = doc.page_count + if total == 0: + return False + with_text = sum(1 for page in doc if page.get_text("text").strip()) + return with_text * 2 > total except Exception: pass return False use_ocr = not _pdf_has_text(pdf_path) + if use_ocr and self.converter_ocr is None: + print(f"{pdf_path} has no text layer but no OCR converter is available; trying without OCR.") + use_ocr = False converter = self.converter_ocr if use_ocr else self.converter_no_ocr ocr_msg = "(OCR enabled)" if use_ocr else "(no OCR)" + if converter is None: + print(f"No docling PDF converter available. Skipping {pdf_path}.") + return [] + print(f"Converting {pdf_path} to Markdown using docling {ocr_msg}...") return self._perform_conversion(pdf_path, converter, ocr_msg) - - def _convert_txt_to_markdown(self, file_path: str) -> List[Tuple[str, Dict[str, Any]]]: + + def _convert_txt_to_markdown(self, file_path: str) -> ConversionResults: """Convert plain text files to markdown by reading content directly.""" print(f"Converting {file_path} (TXT) to Markdown...") try: @@ -110,12 +190,15 @@ def _convert_txt_to_markdown(self, file_path: str) -> List[Tuple[str, Dict[str, print(f"Error processing TXT file {file_path}: {e}") return [] - def _convert_general_to_markdown(self, file_path: str, input_format: InputFormat) -> List[Tuple[str, Dict[str, Any]]]: + def _convert_general_to_markdown(self, file_path: str, input_format: InputFormat) -> ConversionResults: """Convert non-PDF formats using general converter.""" + if self.converter_general is None: + print(f"General docling converter not available. Skipping {file_path}.") + return [] print(f"Converting {file_path} ({input_format.name}) to Markdown using docling...") return self._perform_conversion(file_path, self.converter_general, f"({input_format.name})") - def _perform_conversion(self, file_path: str, converter, format_msg: str) -> List[Tuple[str, Dict[str, Any]]]: + def _perform_conversion(self, file_path: str, converter, format_msg: str) -> ConversionResults: """Perform the actual conversion using the specified converter.""" pages_data = [] try: diff --git a/rag_system/main.py b/rag_system/main.py index a1f50794..cc6b8767 100644 --- a/rag_system/main.py +++ b/rag_system/main.py @@ -1,38 +1,29 @@ import os import json -import sys import argparse from dotenv import load_dotenv # Load environment variables from .env file load_dotenv() -# The sys.path manipulation has been removed to prevent import conflicts. -# This script should be run as a module from the project root, e.g.: +# This module holds the MASTER configuration for the RAG system plus a thin CLI. +# Agent / pipeline construction lives in rag_system/factory.py. +# Run it as a module from the project root, e.g.: # python -m rag_system.main api -from rag_system.agent.loop import Agent -from rag_system.utils.ollama_client import OllamaClient -# Configuration is now defined in this file - no import needed - -# Advanced RAG System Configuration -# ================================== -# This file contains the MASTER configuration for all models used in the RAG system. -# All components should reference these configurations to ensure consistency. - # ============================================================================ -# ๐ŸŽฏ MASTER MODEL CONFIGURATION +# MASTER MODEL CONFIGURATION # ============================================================================ -# All model configurations are centralized here to prevent conflicts +# Every model default below can be overridden with an environment variable. -# LLM Backend Configuration +# LLM Backend Configuration ("ollama" or "watsonx") LLM_BACKEND = os.getenv("LLM_BACKEND", "ollama") # Ollama Models Configuration (for inference via Ollama) OLLAMA_CONFIG = { "host": os.getenv("OLLAMA_HOST", "http://localhost:11434"), - "generation_model": "qwen3:8b", # Main text generation model - "enrichment_model": "qwen3:0.6b", # Lightweight model for routing/enrichment + "generation_model": os.getenv("GENERATION_MODEL", "qwen3.5:9b"), + "enrichment_model": os.getenv("ENRICHMENT_MODEL", "qwen3.5:4b"), } WATSONX_CONFIG = { @@ -40,98 +31,171 @@ "project_id": os.getenv("WATSONX_PROJECT_ID", ""), "url": os.getenv("WATSONX_URL", "https://us-south.ml.cloud.ibm.com"), "generation_model": os.getenv("WATSONX_GENERATION_MODEL", "ibm/granite-13b-chat-v2"), - "enrichment_model": os.getenv("WATSONX_ENRICHMENT_MODEL", "ibm/granite-8b-japanese"), # Lightweight model + "enrichment_model": os.getenv("WATSONX_ENRICHMENT_MODEL", "ibm/granite-8b-japanese"), } -# External Model Configuration (HuggingFace models used directly) +# External Model Configuration (HuggingFace models loaded in-process) +# +# Defaults set at the Phase 1 adoption gate (2026-08-09); the measurements and +# the reasoning are in eval/DECISIONS.md. +# +# embedding_model microsoft/harrier-oss-v1-0.6b (MIT, 1024-dim). Measured +# mixed-corpus first-stage nDCG@10 0.915 vs 0.875 for +# Qwen/Qwen3-Embedding-4B, at ~3x lower latency and ~7x less +# memory. Qwen/Qwen3-Embedding-4B remains a supported option. +# reranker_model Only loaded when reranking is switched on โ€” the "default" +# profile ships with reranker.enabled = False (see below). +# When a user does switch it on, they get the model that +# measured a win on top of this first stage. EXTERNAL_MODELS = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", # HuggingFace embedding model (1024 dims - fresh start) - "reranker_model": "answerdotai/answerai-colbert-small-v1", # ColBERT reranker - "vision_model": "Qwen/Qwen-VL-Chat", # Vision model for multimodal - "fallback_reranker": "BAAI/bge-reranker-base", # Backup reranker + "embedding_model": os.getenv("EMBEDDING_MODEL", "microsoft/harrier-oss-v1-0.6b"), + "reranker_model": os.getenv("RERANKER_MODEL", "Qwen/Qwen3-Reranker-4B"), } # ============================================================================ -# ๐Ÿ”ง PIPELINE CONFIGURATIONS +# PIPELINE CONFIGURATIONS # ============================================================================ PIPELINE_CONFIGS = { "default": { - "description": "Production-ready pipeline with hybrid search, AI reranking, and verification", + "description": "Production-ready pipeline with hybrid search, query decomposition, and verification", "storage": { - "lancedb_uri": "./lancedb", - "text_table_name": "text_pages_v3", - "image_table_name": "image_pages_v3", - "bm25_path": "./index_store/bm25", - "graph_path": "./index_store/graph/knowledge_graph.gml" + "lancedb_uri": os.getenv("LANCEDB_PATH", "./lancedb"), + # v4: vectors are L2-normalized at write and query time (cosine + # ordering). v3 tables hold unnormalized vectors โ€” see + # rag_system/indexing/embedders.py table markers. + "text_table_name": "text_pages_v4" }, "retrieval": { - "retriever": "multivector", "search_type": "hybrid", - "late_chunking": { - "enabled": True, - "table_suffix": "_lc_v3" - }, - "dense": { - "enabled": True, - "weight": 0.7 + # OFF by default since the 2026-08-18 component ablation + # (eval/decisions/component-ablation-2026-08-18.md): removing the + # late-chunk leg measured -2/120 (noise floor, churn 9 down/7 up) + # while the leg doubles vectors written at indexing and adds a + # second table per index. Opt back in per config; the same flag + # governs both the index-time build and the query-time leg. + "latechunk": { + "enabled": False + }, + "dense": { + "enabled": True }, - "bm25": { + # Evidence-sufficiency retry (roadmap 2.1). One conditional second + # retrieval when the first pass found weak evidence; hard cap of one + # extra attempt. The signal is NOT the raw top cosine โ€” that measured + # anti-correlated with success โ€” but the contrast between the best + # candidate and the background of the rest; see + # RetrievalPipeline._dense_evidence_score and eval/decisions/ + # phase2-pipeline.md for the calibration. + "retry": { "enabled": True, - "index_name": "rag_bm25_index" + "min_top_score": 0.12, + "max_attempts": 1 }, - "graph": { + # Full-document escalation (roadmap 4.1). OFF until benchmarked. + # When the evidence-sufficiency retry above has already run and the + # evidence is STILL below threshold, reassemble the top-ranked + # chunk's whole document in chunk_index order and append it to the + # synthesis context, capped at token_budget. One document, no loop. + # min_evidence defaults to retry.min_top_score when omitted. See + # eval/decisions/phase4-escalation-tokens.md. + "document_escalation": { "enabled": False, - "graph_path": "./index_store/graph/knowledge_graph.gml" + "max_documents": 1, + "token_budget": 6000 + }, + # Cross-reference hop (roadmap 4.2). OFF: index-time extraction is + # free and additive, but the query-time hop appends chunks the + # retriever never scored, and it has not been benchmarked yet. See + # eval/decisions/phase4-crossref-prefilter.md. + "crossref_hop": { + "enabled": False, + "max_hops": 1, # referenced documents expanded, no recursion + "chunks_per_hop": 3 + }, + # Overview prefilter (roadmap 4.3). OFF until benchmarked. "boost" + # is the safe mode โ€” it reorders; "restrict" can hide a document. + "overview_prefilter": { + "enabled": False, + "top_documents": 5, + "mode": "boost" # "boost" | "restrict" } }, - # ๐ŸŽฏ EMBEDDING MODEL: Uses HuggingFace Qwen model directly "embedding_model_name": EXTERNAL_MODELS["embedding_model"], - # ๐ŸŽฏ VISION MODEL: For multimodal capabilities - "vision_model_name": EXTERNAL_MODELS["vision_model"], - # ๐ŸŽฏ RERANKER: AI-powered reranking with ColBERT + # Reranking is ON with threshold selection (arm G). The Phase 1 "off by + # default" call (first stage alone nDCG@10 0.915; bge drops it to 0.892; + # Qwen3-Reranker-4B lifts to 0.977 at ~12.7s/query) predates the + # synthesis context budget โ€” back then rank order barely mattered + # because front-truncation fed synthesis the tail of the list anyway. + # Now the budget keeps exactly the top-ranked docs, so ordering AND + # selection decide everything the model reads. min_score keeps only + # candidates the calibrated Qwen scorer marks relevant to at least one + # query (union across sub-queries), instead of a fixed 10. "reranker": { - "enabled": True, - "type": "ai", + "enabled": True, + "model_type": "cross-encoder", "strategy": "rerankers-lib", "model_name": EXTERNAL_MODELS["reranker_model"], - "top_k": 10 + "top_k": 10, + # Qwen scorer only (calibrated P(relevant)): candidates below this + # against every query are dropped, so easy questions send a small, + # clean context instead of a fixed-size one (arm G, 2026-08-14). + "min_score": 0.5, + "min_keep": 3 }, "query_decomposition": { "enabled": True, - "max_sub_queries": 3, - "compose_from_sub_answers": True + # Arm H (2026-08-15): decomposed queries retrieve per-sub-query, + # pool + dedupe the candidates, then ONE source-aware rerank pass + # and ONE synthesis over the union context โ€” replacing N rerank + # passes, N synthesis calls and the compose step (where multi-hop + # facts were measurably lost, arm E). + "compose_from_sub_answers": False, + "pooled_first_stage": True, + # Resolve-only candidate (component ablation 2026-08-18): use the + # decomposer's context resolution but never split. OFF pending its + # own A/B + multiturn gate. + "resolve_only": False }, - "verification": {"enabled": True}, + # OFF by default since the 2026-08-18 component ablation: verdicts were + # byte-identical on all 120 gold rows with it off (it annotates, never + # changes answers) and it costs one utility-model call per query. + "verification": {"enabled": False}, "retrieval_k": 20, "context_window_size": 0, "semantic_cache_threshold": 0.98, - "cache_scope": "global", - # ๐Ÿ”ง Contextual enrichment configuration + "cache_scope": "session", "contextual_enricher": { "enabled": True, "window_size": 1 }, - # ๐Ÿ”ง Indexing configuration "indexing": { "embedding_batch_size": 50, "enrichment_batch_size": 10, - "enable_progress_tracking": True + # Cross-reference extraction (roadmap 4.2, index-time half). Regex + # only, no LLM; writes chunk metadata.crossrefs. Verified inert for + # retrieval: text and vector columns are bit-identical with it on. + "extract_crossrefs": True } }, "fast": { "description": "Speed-optimized pipeline with minimal overhead", "storage": { - "lancedb_uri": "./lancedb", - "text_table_name": "text_pages_v3", - "image_table_name": "image_pages_v3", - "bm25_path": "./index_store/bm25" + "lancedb_uri": os.getenv("LANCEDB_PATH", "./lancedb"), + "text_table_name": "text_pages_v4" }, "retrieval": { - "retriever": "multivector", "search_type": "vector_only", - "late_chunking": {"enabled": False}, - "dense": {"enabled": True} + "latechunk": {"enabled": False}, + "dense": {"enabled": True}, + # Off in `fast`: the retry costs one enrichment-model round-trip plus + # a second retrieval, which is exactly what this profile exists to avoid. + "retry": {"enabled": False}, + # Phase-4 flags (roadmap 4.1โ€“4.3): all off in `fast` for the same + # reason โ€” this profile exists to avoid extra work per query. + "document_escalation": {"enabled": False}, + "crossref_hop": {"enabled": False}, + "overview_prefilter": {"enabled": False} }, "embedding_model_name": EXTERNAL_MODELS["embedding_model"], "reranker": {"enabled": False}, @@ -139,231 +203,156 @@ "verification": {"enabled": False}, "retrieval_k": 10, "context_window_size": 0, - # ๐Ÿ”ง Contextual enrichment (disabled for speed) + "semantic_cache_threshold": 0.98, + "cache_scope": "session", "contextual_enricher": { "enabled": False, "window_size": 1 }, - # ๐Ÿ”ง Indexing configuration "indexing": { "embedding_batch_size": 100, "enrichment_batch_size": 50, - "enable_progress_tracking": False + # Costs nothing even in `fast` (regex over text already in memory). + "extract_crossrefs": True } - }, - "bm25": { - "enabled": True, - "index_name": "rag_bm25_index" - }, - "graph_rag": { - "enabled": False, # Keep disabled for now unless specified } } # ============================================================================ -# ๐Ÿญ FACTORY FUNCTIONS +# CLI # ============================================================================ -def get_agent(mode: str = "default") -> Agent: - """ - Factory function to get an instance of the RAG agent based on the specified mode. - - Args: - mode: Configuration mode ("default", "fast") - - Returns: - Configured Agent instance - """ - load_dotenv() - - # Initialize the appropriate LLM client based on backend configuration - if LLM_BACKEND.lower() == "watsonx": - from rag_system.utils.watsonx_client import WatsonXClient - - if not WATSONX_CONFIG["api_key"] or not WATSONX_CONFIG["project_id"]: - raise ValueError( - "Watson X configuration incomplete. Please set WATSONX_API_KEY and WATSONX_PROJECT_ID " - "environment variables." - ) - - llm_client = WatsonXClient( - api_key=WATSONX_CONFIG["api_key"], - project_id=WATSONX_CONFIG["project_id"], - url=WATSONX_CONFIG["url"] - ) - llm_config = WATSONX_CONFIG - print(f"๐Ÿ”ง Using Watson X backend with granite models") - else: - llm_client = OllamaClient(host=OLLAMA_CONFIG["host"]) - llm_config = OLLAMA_CONFIG - print(f"๐Ÿ”ง Using Ollama backend") - - # Get the configuration for the specified mode - config = PIPELINE_CONFIGS.get(mode, PIPELINE_CONFIGS['default']) - - agent = Agent( - pipeline_configs=config, - llm_client=llm_client, - ollama_config=llm_config +SUPPORTED_DOCUMENT_EXTENSIONS = (".pdf", ".docx", ".html", ".htm", ".md", ".txt") + + +def _collect_file_paths(path: str) -> list[str]: + """Expand a file or directory argument into a list of indexable file paths.""" + if os.path.isfile(path): + return [os.path.abspath(path)] + + if not os.path.isdir(path): + raise FileNotFoundError(f"No such file or directory: {path}") + + collected = [] + for root, _dirs, files in os.walk(path): + for name in sorted(files): + if name.lower().endswith(SUPPORTED_DOCUMENT_EXTENSIONS): + collected.append(os.path.join(root, name)) + return collected + + +def main() -> int: + parser = argparse.ArgumentParser( + prog="python -m rag_system.main", + description="localGPT RAG system: indexing, one-shot chat, and API server." ) - return agent - -def validate_model_config(): - """ - Validates the model configuration for consistency and availability. - - Raises: - ValueError: If configuration conflicts are detected - """ - print("๐Ÿ” Validating model configuration...") - - # Check for embedding model consistency - default_embedding = PIPELINE_CONFIGS["default"]["embedding_model_name"] - external_embedding = EXTERNAL_MODELS["embedding_model"] - - if default_embedding != external_embedding: - raise ValueError(f"Embedding model mismatch: {default_embedding} != {external_embedding}") - - # Check reranker configuration - default_reranker = PIPELINE_CONFIGS["default"]["reranker"]["model_name"] - external_reranker = EXTERNAL_MODELS["reranker_model"] - - if default_reranker != external_reranker: - raise ValueError(f"Reranker model mismatch: {default_reranker} != {external_reranker}") - - print("โœ… Model configuration validation passed!") - - return True + subparsers = parser.add_subparsers(dest="command", required=True) -# ============================================================================ -# ๐Ÿš€ UTILITY FUNCTIONS -# ============================================================================ + modes = sorted(PIPELINE_CONFIGS) -def run_indexing(docs_path: str, config_mode: str = "default"): - """Runs the indexing pipeline for the specified documents.""" - print(f"๐Ÿ“š Starting indexing for documents in: {docs_path}") - validate_model_config() - - # Local import to avoid circular dependencies - from rag_system.pipelines.indexing_pipeline import IndexingPipeline - - # Get the appropriate indexing pipeline from the factory - indexing_pipeline = IndexingPipeline(PIPELINE_CONFIGS[config_mode]) - - # Find all PDF files in the directory - pdf_files = [os.path.join(docs_path, f) for f in os.listdir(docs_path) if f.endswith(".pdf")] - - if not pdf_files: - print("No PDF files found to index.") - return - - # Process all documents through the pipeline - indexing_pipeline.process_documents(pdf_files) - print("โœ… Indexing complete.") - -def run_chat(query: str): - """ - Runs the agentic RAG pipeline for a given query. - Returns the result as a JSON string. - """ - try: - validate_model_config() - ollama_client = OllamaClient(OLLAMA_CONFIG["host"]) - except ConnectionError as e: - print(e) - return json.dumps({"error": str(e)}, indent=2) - except ValueError as e: - print(f"Configuration Error: {e}") - return json.dumps({"error": f"Configuration Error: {e}"}, indent=2) - - agent = Agent(PIPELINE_CONFIGS['default'], ollama_client, OLLAMA_CONFIG) - result = agent.run(query) - return json.dumps(result, indent=2, ensure_ascii=False) - -def show_graph(): - """ - Loads and displays the knowledge graph. - """ - import networkx as nx - import matplotlib.pyplot as plt - - graph_path = PIPELINE_CONFIGS["indexing"]["graph_path"] - if not os.path.exists(graph_path): - print("Knowledge graph not found. Please run the 'index' command first.") - return - - G = nx.read_gml(graph_path) - print("--- Knowledge Graph ---") - print("Nodes:", G.nodes(data=True)) - print("Edges:", G.edges(data=True)) - print("---------------------") - - # Optional: Visualize the graph - try: - pos = nx.spring_layout(G) - nx.draw(G, pos, with_labels=True, node_size=2000, node_color="skyblue", font_size=10, font_weight="bold") - edge_labels = nx.get_edge_attributes(G, 'label') - nx.draw_networkx_edge_labels(G, pos, edge_labels=edge_labels) - plt.title("Knowledge Graph Visualization") - plt.show() - except Exception as e: - print(f"\nCould not visualize the graph. Matplotlib might not be installed or configured for your environment.") - print(f"Error: {e}") - -def run_api_server(): - """Starts the advanced RAG API server.""" - from rag_system.api_server import start_server - start_server() - -def main(): - if len(sys.argv) < 2: - print("Usage: python main.py [index|chat|show_graph|api] [query]") - return - - command = sys.argv[1] - if command == "index": - # Allow passing file paths from the command line - files = sys.argv[2:] if len(sys.argv) > 2 else None - run_indexing(files) - elif command == "chat": - if len(sys.argv) < 3: - print("Usage: python main.py chat <query>") - return - query = " ".join(sys.argv[2:]) - # ๐Ÿ†• Print the result for command-line usage - print(run_chat(query)) - elif command == "show_graph": - show_graph() - elif command == "api": - run_api_server() - else: - print(f"Unknown command: {command}") + index_parser = subparsers.add_parser("index", help="Index a document or a directory of documents.") + index_parser.add_argument("path", help="File or directory to index.") + index_parser.add_argument("--mode", default="default", choices=modes, help="Pipeline profile to use.") -if __name__ == "__main__": - # This allows running the script from the command line to index documents. - parser = argparse.ArgumentParser(description="Main entry point for the RAG system.") - parser.add_argument( - '--index', - type=str, - help='Path to the directory containing documents to index.' + chat_parser = subparsers.add_parser("chat", help="Answer a single query and print the JSON result.") + chat_parser.add_argument("query", help="The question to ask.") + chat_parser.add_argument("--mode", default="default", choices=modes, help="Pipeline profile to use.") + chat_parser.add_argument( + "--filters", default=None, + help='Metadata filter as JSON (roadmap 4.4), e.g. \'{"document_id": "nda.pdf"}\' ' + 'or \'{"document_name": {"contains": "nda"}, "chunk_index": {"lte": 4}}\'.' ) - parser.add_argument( - '--config', - type=str, - default='default', - help='The configuration profile to use (e.g., "default", "fast").' + + # Ephemeral "ask a folder" mode (roadmap 4.6). + ask_parser = subparsers.add_parser( + "ask", + help="Index a folder into a throwaway index, answer, then delete it." ) + ask_parser.add_argument("path", help="Folder (or single file) to ask about.") + ask_parser.add_argument("questions", nargs="*", help="One or more questions.") + ask_parser.add_argument("--mode", default="fast", choices=modes, + help="Pipeline profile to use (default: fast).") + ask_parser.add_argument("--interactive", action="store_true", + help="Keep asking follow-up questions against the same temp index.") + ask_parser.add_argument("--agent", action="store_true", + help="Answer through the full agent loop (decomposition, " + "verification) instead of the retrieval pipeline alone.") + ask_parser.add_argument("--filters", default=None, + help="Metadata filter as JSON (see 'chat --filters').") + ask_parser.add_argument("--keep", action="store_true", + help="Do not delete the temporary index (debugging).") + + api_parser = subparsers.add_parser("api", help="Start the RAG API server.") + api_parser.add_argument("--port", type=int, default=8001, help="Port to listen on.") args = parser.parse_args() - # Load environment variables - load_dotenv() - - if args.index: - run_indexing(args.index, args.config) - else: - # This is where you might start a server or interactive session - print("No action specified. Use --index to process documents.") - # Example of how to get an agent instance - # agent = get_agent(args.config) - # print(f"Agent loaded with '{args.config}' config.") + def _parse_filters(raw): + """Parse and validate --filters. Returns (compiled_or_None, exit_code_or_None).""" + if not raw: + return None, None + from rag_system.retrieval.filters import FilterError, compile_filters + try: + return compile_filters(json.loads(raw)), None + except json.JSONDecodeError as e: + print(f"โŒ --filters is not valid JSON: {e}") + return None, 2 + except FilterError as e: + print(f"โŒ Invalid --filters: {e}") + return None, 2 + + if args.command == "index": + from rag_system.factory import get_indexing_pipeline + + try: + file_paths = _collect_file_paths(args.path) + except FileNotFoundError as e: + print(f"โŒ {e}") + return 1 + + if not file_paths: + print(f"No indexable documents found in {args.path} " + f"(supported: {', '.join(SUPPORTED_DOCUMENT_EXTENSIONS)}).") + return 1 + + print(f"๐Ÿ“š Indexing {len(file_paths)} file(s) with the '{args.mode}' profile...") + get_indexing_pipeline(args.mode).run(file_paths) + print("โœ… Indexing complete.") + return 0 + + if args.command == "chat": + from rag_system.factory import get_agent + + filters, error = _parse_filters(args.filters) + if error: + return error + + result = get_agent(args.mode).run(args.query, filters=filters) + print(json.dumps(result, indent=2, ensure_ascii=False)) + return 0 + + if args.command == "ask": + from rag_system.ask_folder import ask_folder + + filters, error = _parse_filters(args.filters) + if error: + return error + + return ask_folder( + args.path, args.questions, mode=args.mode, + interactive=args.interactive, use_agent=args.agent, + filters=filters, keep=args.keep, + ) + + if args.command == "api": + from rag_system.api_server import start_server + + start_server(port=args.port) + return 0 + + parser.error(f"Unknown command: {args.command}") + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/rag_system/pipelines/indexing_pipeline.py b/rag_system/pipelines/indexing_pipeline.py index 9fc61e7e..d53e6612 100644 --- a/rag_system/pipelines/indexing_pipeline.py +++ b/rag_system/pipelines/indexing_pipeline.py @@ -1,15 +1,22 @@ from typing import List, Dict, Any +import hashlib import os -import networkx as nx from rag_system.ingestion.document_converter import DocumentConverter from rag_system.ingestion.chunking import MarkdownRecursiveChunker from rag_system.indexing.representations import EmbeddingGenerator, select_embedder from rag_system.indexing.embedders import LanceDBManager, VectorIndexer -from rag_system.indexing.graph_extractor import GraphExtractor from rag_system.utils.ollama_client import OllamaClient from rag_system.indexing.contextualizer import ContextualEnricher +from rag_system.indexing.crossref import annotate_chunks from rag_system.indexing.overview_builder import OverviewBuilder + +def _default_embedding_model() -> str: + """The single source of truth for the embedding model default.""" + from rag_system.main import EXTERNAL_MODELS + return EXTERNAL_MODELS["embedding_model"] + + class IndexingPipeline: def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_config: Dict[str, str]): self.config = config @@ -18,35 +25,38 @@ def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_c self.document_converter = DocumentConverter() # Chunker selection: docling (token-based) or legacy (character-based) chunker_mode = config.get("chunker_mode", "docling") - - # ๐Ÿ”ง Get chunking configuration from frontend parameters + + self.embedding_model_name = config.get("embedding_model_name") or _default_embedding_model() + + # Chunk size is the token budget per chunk for both chunkers. chunking_config = config.get("chunking", {}) - chunk_size = chunking_config.get("chunk_size", config.get("chunk_size", 1500)) - chunk_overlap = chunking_config.get("chunk_overlap", config.get("chunk_overlap", 200)) - - print(f"๐Ÿ”ง CHUNKING CONFIG: Size: {chunk_size}, Overlap: {chunk_overlap}, Mode: {chunker_mode}") - + chunk_size = chunking_config.get( + "chunk_size", config.get("chunk_size", config.get("max_tokens", 1500)) + ) + + print(f"๐Ÿ”ง CHUNKING CONFIG: Size: {chunk_size}, Mode: {chunker_mode}") + if chunker_mode == "docling": try: from rag_system.ingestion.docling_chunker import DoclingChunker self.chunker = DoclingChunker( - max_tokens=config.get("max_tokens", chunk_size), + max_tokens=chunk_size, overlap=config.get("overlap_sentences", 1), - tokenizer_model=config.get("embedding_model_name", "qwen3-embedding-0.6b"), + tokenizer_model=self.embedding_model_name, ) print("๐Ÿช„ Using DoclingChunker for high-recall sentence packing.") except Exception as e: print(f"โš ๏ธ Failed to initialise DoclingChunker: {e}. Falling back to legacy chunker.") self.chunker = MarkdownRecursiveChunker( max_chunk_size=chunk_size, - min_chunk_size=min(chunk_overlap, chunk_size // 4), # Sensible minimum - tokenizer_model=config.get("embedding_model_name", "Qwen/Qwen3-Embedding-0.6B") + min_chunk_size=max(1, chunk_size // 4), + tokenizer_model=self.embedding_model_name, ) else: self.chunker = MarkdownRecursiveChunker( max_chunk_size=chunk_size, - min_chunk_size=min(chunk_overlap, chunk_size // 4), # Sensible minimum - tokenizer_model=config.get("embedding_model_name", "Qwen/Qwen3-Embedding-0.6B") + min_chunk_size=max(1, chunk_size // 4), + tokenizer_model=self.embedding_model_name, ) retriever_configs = self.config.get("retrievers") or self.config.get("retrieval", {}) @@ -56,7 +66,13 @@ def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_c indexing_config = self.config.get("indexing", {}) self.embedding_batch_size = indexing_config.get("embedding_batch_size", 50) self.enrichment_batch_size = indexing_config.get("enrichment_batch_size", 10) - self.enable_progress_tracking = indexing_config.get("enable_progress_tracking", True) + + # Cross-reference extraction (roadmap item 4.2). On by default: it is a + # few regexes over text already in memory, adds no LLM call and no second + # pass, and only writes chunk metadata โ€” an index built with it is + # byte-identical on the `text` and `vector` columns, so nothing + # downstream changes until the query-time hop flag is switched on. + self.extract_crossrefs = bool(indexing_config.get("extract_crossrefs", True)) # Treat dense retrieval as enabled by default unless explicitly disabled dense_cfg = retriever_configs.setdefault("dense", {}) @@ -76,7 +92,7 @@ def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_c self.lancedb_manager = LanceDBManager(db_path=db_path) self.vector_indexer = VectorIndexer(self.lancedb_manager) embedding_model = select_embedder( - self.config.get("embedding_model_name", "BAAI/bge-small-en-v1.5"), + self.embedding_model_name, self.ollama_config.get("host") if isinstance(self.ollama_config, dict) else None, ) self.embedding_generator = EmbeddingGenerator( @@ -84,46 +100,60 @@ def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_c batch_size=self.embedding_batch_size ) - if retriever_configs.get("graph", {}).get("enabled"): - self.graph_extractor = GraphExtractor( - llm_client=self.llm_client, - llm_model=self.ollama_config["generation_model"] - ) - - if self.config.get("contextual_enricher", {}).get("enabled"): - # ๐Ÿ”ง Use frontend enrich_model parameter if provided + enricher_config = self.config.get("contextual_enricher", {}) + self.enricher_enabled = bool(enricher_config.get("enabled", False)) + self.enricher_window_size = enricher_config.get("window_size", 1) + self.contextual_enricher = None + if self.enricher_enabled: enrichment_model = ( - self.config.get("enrich_model") or # Frontend parameter + self.config.get("enrich_model") or # Per-request override self.config.get("enrichment_model_name") or # Alternative config key - self.ollama_config.get("enrichment_model") or # Default from ollama config + self.ollama_config.get("enrichment_model") or # Default from llm config self.ollama_config["generation_model"] # Final fallback ) print(f"๐Ÿ”ง ENRICHMENT MODEL: Using '{enrichment_model}' for contextual enrichment") - + self.contextual_enricher = ContextualEnricher( llm_client=self.llm_client, llm_model=enrichment_model, batch_size=self.enrichment_batch_size ) - # Overview builder always enabled for triage routing - ov_path = self.config.get("overview_path") - self.overview_builder = OverviewBuilder( - llm_client=self.llm_client, - model=self.config.get("overview_model_name", self.ollama_config.get("enrichment_model", "qwen3:0.6b")), - first_n_chunks=self.config.get("overview_first_n_chunks", 5), - out_path=ov_path if ov_path else None, - ) + # Document overviews feed the triage router; on by default. + overview_config = self.config.get("overview", {}) + self.overview_builder = None + # Embedded-overview sidecar for the query-time overview prefilter + # (roadmap item 4.3). One embedding per *document*, so it is negligible + # next to the per-chunk pass that already runs. + self.embed_overviews = bool(overview_config.get("embed", True)) + if overview_config.get("enabled", True): + self.overview_builder = OverviewBuilder( + llm_client=self.llm_client, + model=( + self.config.get("overview_model_name") + or overview_config.get("model") + or self.ollama_config.get("enrichment_model") + or self.ollama_config["generation_model"] + ), + first_n_chunks=self.config.get( + "overview_first_n_chunks", overview_config.get("max_chunks", 5) + ), + out_path=self.config.get("overview_path") or None, + ) # ------------------------------------------------------------------ # Late-Chunk encoder initialisation (optional) # ------------------------------------------------------------------ - self.latechunk_enabled = retriever_configs.get("latechunk", {}).get("enabled", False) + self.latechunk_cfg = ( + retriever_configs.get("latechunk") + or retriever_configs.get("late_chunking") + or {} + ) + self.latechunk_enabled = bool(self.latechunk_cfg.get("enabled", False)) if self.latechunk_enabled: try: from rag_system.indexing.latechunk import LateChunkEncoder - self.latechunk_cfg = retriever_configs["latechunk"] - self.latechunk_encoder = LateChunkEncoder(model_name=self.config.get("embedding_model_name", "qwen3-embedding-0.6b")) + self.latechunk_encoder = LateChunkEncoder(model_name=self.embedding_model_name) except Exception as e: print(f"โš ๏ธ Failed to initialise LateChunkEncoder: {e}. Disabling latechunk retrieval.") self.latechunk_enabled = False @@ -151,10 +181,18 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non doc_chunks_map = {} with timer("Document Processing & Chunking"): file_tracker = ProgressTracker(len(file_paths), "Document Processing") - + seen_document_ids = set() + for file_path in file_paths: try: document_id = os.path.basename(file_path) + if document_id in seen_document_ids: + # Two files share a basename: disambiguate with a + # deterministic suffix of the full path so their + # document_id / chunk_id namespaces can't collide. + suffix = hashlib.sha1(file_path.encode("utf-8")).hexdigest()[:8] + document_id = f"{document_id}#{suffix}" + seen_document_ids.add(document_id) print(f"Processing: {document_id}") pages_data = self.document_converter.convert_to_markdown(file_path) @@ -179,11 +217,12 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non chunk['metadata']['chunk_index'] = i # Build and persist document overview (non-blocking errors) - try: - self.overview_builder.build_and_store(document_id, file_chunks) - except Exception as e: - print(f" โš ๏ธ Failed to create overview for {document_id}: {e}") - + if self.overview_builder is not None: + try: + self.overview_builder.build_and_store(document_id, file_chunks) + except Exception as e: + print(f" โš ๏ธ Failed to create overview for {document_id}: {e}") + all_chunks.extend(file_chunks) doc_chunks_map[document_id] = file_chunks # save for late-chunk step print(f" Generated {len(file_chunks)} chunks from {document_id}") @@ -197,62 +236,59 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non file_tracker.finish() if not all_chunks: - print("No text chunks were generated. Skipping indexing.") - return + raise RuntimeError( + "No text chunks were generated from the supplied documents โ€” " + "conversion or chunking failed for every file. Check the server " + "log for per-file conversion errors; nothing was indexed." + ) print(f"\nโœ… Generated {len(all_chunks)} text chunks total.") memory_mb = estimate_memory_usage(all_chunks) print(f"๐Ÿ“Š Estimated memory usage: {memory_mb:.1f}MB") retriever_configs = self.config.get("retrievers") or self.config.get("retrieval", {}) + table_name = self._text_table_name(retriever_configs) + + # Step 1b: Cross-reference extraction (roadmap item 4.2) + # Runs on the ORIGINAL chunk text, before contextual enrichment + # rewrites it โ€” an enriched chunk carries an LLM-written preamble + # that can invent or drop a reference. + if self.extract_crossrefs: + with timer("Cross-reference extraction"): + known = self._existing_document_ids(table_name) + stats = annotate_chunks(doc_chunks_map, known_documents=known) + print( + f"๐Ÿ”— Cross-references: {stats['refs']} reference(s) in " + f"{stats['chunks_with_refs']} chunk(s); {stats['resolved']} resolved " + f"to {stats['documents_linked']} document(s)." + ) - # Step 3: Optional Contextual Enrichment (before indexing for consistency) - enricher_config = self.config.get("contextual_enricher", {}) - enricher_enabled = enricher_config.get("enabled", False) - - print(f"\n๐Ÿ” CONTEXTUAL ENRICHMENT DEBUG:") - print(f" Config present: {bool(enricher_config)}") - print(f" Enabled: {enricher_enabled}") - print(f" Has enricher object: {hasattr(self, 'contextual_enricher')}") - - if hasattr(self, 'contextual_enricher') and enricher_enabled: + # Step 2: Optional Contextual Enrichment (before indexing for consistency) + if self.contextual_enricher is not None: with timer("Contextual Enrichment"): - window_size = enricher_config.get("window_size", 1) - print(f"\n๐Ÿš€ CONTEXTUAL ENRICHMENT ACTIVE!") - print(f" Window size: {window_size}") - print(f" Model: {self.contextual_enricher.llm_model}") - print(f" Batch size: {self.contextual_enricher.batch_size}") - print(f" Processing {len(all_chunks)} chunks...") - - # Show before/after example - if all_chunks: - print(f" Example BEFORE: '{all_chunks[0]['text'][:100]}...'") - + print( + f"\n๐Ÿš€ Contextual enrichment: model={self.contextual_enricher.llm_model}, " + f"window={self.enricher_window_size}, batch={self.contextual_enricher.batch_size}, " + f"chunks={len(all_chunks)}" + ) # This modifies the 'text' field in each chunk dictionary - all_chunks = self.contextual_enricher.enrich_chunks(all_chunks, window_size=window_size) - - if all_chunks: - print(f" Example AFTER: '{all_chunks[0]['text'][:100]}...'") - + all_chunks = self.contextual_enricher.enrich_chunks( + all_chunks, window_size=self.enricher_window_size + ) print(f"โœ… Enriched {len(all_chunks)} chunks with context for indexing.") else: - print(f"โš ๏ธ CONTEXTUAL ENRICHMENT SKIPPED:") - if not hasattr(self, 'contextual_enricher'): - print(f" Reason: No enricher object (config enabled={enricher_enabled})") - elif not enricher_enabled: - print(f" Reason: Disabled in config") - print(f" Chunks will be indexed without contextual enrichment.") - - # Step 4: Create BM25 Index from enriched chunks (for consistency with vector index) + print("\nโ„น๏ธ Contextual enrichment disabled; indexing chunks as-is.") + + # Step 3: Embed chunks into LanceDB and build the native FTS index if hasattr(self, 'vector_indexer') and hasattr(self, 'embedding_generator'): with timer("Vector Embedding & Indexing"): - table_name = self.config["storage"].get("text_table_name") or retriever_configs.get("dense", {}).get("lancedb_table_name", "default_text_table") - print(f"\n--- Generating embeddings with {self.config.get('embedding_model_name')} ---") + print(f"\n--- Generating embeddings with {self.embedding_model_name} ---") embeddings = self.embedding_generator.generate(all_chunks) print(f"\n--- Indexing {len(embeddings)} vectors into LanceDB table: {table_name} ---") - self.vector_indexer.index(table_name, all_chunks, embeddings) + self.vector_indexer.index(table_name, all_chunks, embeddings, + embedding_model=self.embedding_model_name) print("โœ… Vector embeddings indexed successfully") # Create FTS index on the 'text' field after adding data @@ -282,7 +318,9 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non # --------------------------------------------------- if self.latechunk_enabled: with timer("Late-Chunk Embedding & Indexing"): - lc_table_name = self.latechunk_cfg.get("lancedb_table_name", f"{table_name}_lc") + lc_table_name = self.latechunk_cfg.get("lancedb_table_name") or ( + f"{table_name}{self.latechunk_cfg.get('table_suffix', '_lc')}" + ) print(f"\n--- Generating late-chunk embeddings (table={lc_table_name}) ---") total_lc_vecs = 0 @@ -313,46 +351,91 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non print(f"โš ๏ธ Mismatch LC vecs ({len(lc_vecs)}) vs chunks ({len(doc_chunks)}) for {doc_id}. Skipping.") continue - self.vector_indexer.index(lc_table_name, doc_chunks, lc_vecs) + self.vector_indexer.index(lc_table_name, doc_chunks, lc_vecs, + embedding_model=self.embedding_model_name) total_lc_vecs += len(lc_vecs) print(f"โœ… Late-chunk vectors indexed: {total_lc_vecs}") - - # Step 6: Knowledge Graph Extraction (Optional) - if hasattr(self, 'graph_extractor'): - with timer("Knowledge Graph Extraction"): - graph_path = retriever_configs.get("graph", {}).get("graph_path", "./index_store/graph/default_graph.gml") - print(f"\n--- Building and saving knowledge graph to: {graph_path} ---") - - graph_data = self.graph_extractor.extract(all_chunks) - G = nx.DiGraph() - for entity in graph_data['entities']: - G.add_node(entity['id'], type=entity.get('type', 'Unknown'), properties=entity.get('properties', {})) - for rel in graph_data['relationships']: - G.add_edge(rel['source'], rel['target'], label=rel['label']) - - os.makedirs(os.path.dirname(graph_path), exist_ok=True) - nx.write_gml(G, graph_path) - print(f"โœ… Knowledge graph saved successfully.") - + + # The late-chunk table needs its own FTS index: the + # base-table index above does not cover it, and + # without one the hybrid retriever's FTS leg fails + # on this table and silently degrades to dense-only + # (retrievers.py logs "FTS leg failed"). + if total_lc_vecs: + try: + lc_tbl = self.lancedb_manager.get_table(lc_table_name) + lc_indices = [idx.name for idx in lc_tbl.list_indices()] + if not any(n in lc_indices for n in ("text_idx", "fts_text")): + lc_tbl.create_fts_index("text", use_tantivy=False, replace=False) + print(f"โœ… FTS index created on late-chunk table '{lc_table_name}'.") + except Exception as e: + print(f"โŒ Failed to create/verify FTS index on '{lc_table_name}': {e}") + + # Step 4: Embedded-overview sidecar (roadmap item 4.3) + if (self.overview_builder is not None and self.embed_overviews + and hasattr(self, "embedding_generator")): + with timer("Overview Embedding"): + try: + n = self.overview_builder.embed_and_store_vectors( + self.embedding_generator.model, + embedding_model=self.embedding_model_name, + ) + if n: + print(f"๐Ÿงญ Embedded {n} document overview(s) โ†’ " + f"{self.overview_builder.vectors_path}") + except Exception as e: + # A missing sidecar only disables an off-by-default + # query-time feature; it must never fail an index build. + print(f"โš ๏ธ Failed to embed document overviews: {e}") + print("\n--- โœ… Indexing Complete ---") self._print_final_statistics(len(file_paths), len(all_chunks)) + def _text_table_name(self, retriever_configs: Dict[str, Any]) -> str: + return ( + self.config["storage"].get("text_table_name") + or retriever_configs.get("dense", {}).get("lancedb_table_name", "default_text_table") + ) + + def _existing_document_ids(self, table_name: str) -> List[str]: + """Document ids already in the target table, for cross-reference resolution. + + An incremental add should still be able to resolve "Exhibit B" to a + document indexed last week. Best effort only: any failure here just means + references resolve against the current batch alone. + """ + if not table_name or not hasattr(self, "lancedb_manager"): + return [] + try: + db = self.lancedb_manager.db + if hasattr(db, "table_names") and table_name not in db.table_names(): + return [] + tbl = self.lancedb_manager.get_table(table_name) + arrow = tbl.to_lance().to_table(columns=["document_id"]) + return sorted({d for d in arrow.column("document_id").to_pylist() if d}) + except Exception as e: + print(f"โ„น๏ธ Cross-reference resolution limited to this batch ({e}).") + return [] + def _print_final_statistics(self, num_files: int, num_chunks: int): """Print final indexing statistics""" print(f"\n๐Ÿ“ˆ Final Statistics:") print(f" Files processed: {num_files}") print(f" Chunks generated: {num_chunks}") - print(f" Average chunks per file: {num_chunks/num_files:.1f}") - + if num_files: + print(f" Average chunks per file: {num_chunks/num_files:.1f}") + # Component status components = [] - if hasattr(self, 'contextual_enricher'): + if self.contextual_enricher is not None: components.append("โœ… Contextual Enrichment") if hasattr(self, 'vector_indexer'): components.append("โœ… Vector & FTS Index") - if hasattr(self, 'graph_extractor'): - components.append("โœ… Knowledge Graph") - + if self.latechunk_enabled: + components.append("โœ… Late Chunking") + if self.overview_builder is not None: + components.append("โœ… Document Overviews") + print(f" Components: {', '.join(components)}") print(f" Batch sizes: Embeddings={self.embedding_batch_size}, Enrichment={self.enrichment_batch_size}") diff --git a/rag_system/pipelines/retrieval_pipeline.py b/rag_system/pipelines/retrieval_pipeline.py index f2151273..e6513adf 100644 --- a/rag_system/pipelines/retrieval_pipeline.py +++ b/rag_system/pipelines/retrieval_pipeline.py @@ -1,32 +1,39 @@ -import pymupdf -from typing import List, Dict, Any, Tuple, Optional -from PIL import Image +from typing import List, Dict, Any, Optional import concurrent.futures +import contextlib +import threading import time import json -import lancedb import logging import math +import os import numpy as np from threading import Lock from rag_system.utils.ollama_client import OllamaClient -from rag_system.retrieval.retrievers import MultiVectorRetriever, GraphRetriever -from rag_system.indexing.multimodal import LocalVisionModel -from rag_system.indexing.representations import select_embedder -from rag_system.indexing.embedders import LanceDBManager -from rag_system.rerankers.reranker import QwenReranker +from rag_system.retrieval.filters import CompiledFilter, combine, compile_filters +from rag_system.retrieval.retrievers import MultiVectorRetriever +from rag_system.indexing.representations import default_query_instruction, select_embedder +from rag_system.indexing.embedders import ( + EmbedderMismatchError, + LanceDBManager, + assert_embedder_matches, + l2_normalize, + read_table_marker, +) +from rag_system.indexing.overview_builder import load_overview_vectors, overview_vectors_path +from rag_system.rerankers.reranker import CrossEncoderReranker, QwenRerankerScorer, is_qwen3_reranker from rag_system.rerankers.sentence_pruner import SentencePruner -# from rag_system.indexing.chunk_store import ChunkStore -import os -from PIL import Image +# Reciprocal-rank-fusion constant, kept identical to the retriever's so a fused +# ordering produced here is comparable with one produced there. +_RRF_K = 60 # --------------------------------------------------------------------------- # Thread-safety helpers # --------------------------------------------------------------------------- -# 1. ColBERT (via `rerankers` lib) is not thread-safe. We protect the actual +# 1. The `rerankers` lib backends are not thread-safe. We protect the actual # `.rank()` call with `_rerank_lock`. _rerank_lock: Lock = Lock() @@ -42,27 +49,140 @@ class RetrievalPipeline: """ - Orchestrates the state-of-the-art multimodal RAG pipeline. + Orchestrates retrieval, reranking, context expansion, pruning and synthesis. """ def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_config: Dict[str, Any]): self.config = config self.ollama_config = ollama_config self.ollama_client = ollama_client - - # Support both legacy "retrievers" key and newer "retrieval" key - self.retriever_configs = self.config.get("retrievers") or self.config.get("retrieval", {}) + self.storage_config = self.config["storage"] - + # Defer initialization to just-in-time methods self.db_manager = None self.text_embedder = None self.dense_retriever = None - self.bm25_retriever = None - # Use a private attribute to avoid clashing with the public property - self._graph_retriever = None - self.reranker = None + # Set once dense-retriever construction has failed: without it every + # query retried the failing constructor and reprinted the error. + self._dense_retriever_error = None self.ai_reranker = None + # Overview prefilter (roadmap item 4.3): the sidecar is read once per + # pipeline and its absence is reported once, not per query. + self._overview_vectors_loaded = False + self._overview_vectors_data = None + + # Metadata filter (roadmap item 4.4). Thread-local, because the agent + # runs sub-queries of one user question through this same pipeline + # object in parallel and one sub-query's filter must not leak into + # another's search. Empty unless a caller opened a filter scope. + self._filter_local = threading.local() + + def _retriever_config(self, name: str, *aliases: str) -> Dict[str, Any]: + """Look a retriever sub-config up in "retrievers" then "retrieval". + + Resolved on every access instead of cached in ``__init__`` so runtime + overrides written by the API land in the pipeline. + """ + for container_key in ("retrievers", "retrieval"): + container = self.config.get(container_key) or {} + for key in (name, *aliases): + if container.get(key): + return container[key] + return {} + + # ------------------------------------------------------------------ + # Metadata filter scope (roadmap item 4.4) + # ------------------------------------------------------------------ + + def active_filter(self) -> Optional[CompiledFilter]: + """The compiled filter in force on *this thread*, or None.""" + return getattr(self._filter_local, "compiled", None) + + @contextlib.contextmanager + def filter_scope(self, compiled: Optional[CompiledFilter]): + """Make *compiled* the active filter for the duration of the block. + + ``run()`` opens this scope rather than threading a ``filters`` argument + down through every internal call, for one concrete reason: + ``EscalatingRetrievalPipeline`` (roadmap 4.1) overrides + ``retrieve_candidates`` with a fixed four-argument signature and calls + ``super()`` positionally. Adding a parameter that ``run()`` had to pass + would break that subclass. A thread-local scope is invisible to it, and + the agent's parallel sub-query fan-out enters ``run()`` **inside** each + worker thread, so each worker sets its own. + """ + previous = getattr(self._filter_local, "compiled", None) + self._filter_local.compiled = compiled + try: + yield compiled + finally: + self._filter_local.compiled = previous + + def _retrieval_mode(self) -> str: + """Query-time search mode: "hybrid" (default), "vector_only" or "fts_only".""" + retrieval_cfg = self.config.get("retrieval") or {} + return retrieval_cfg.get("search_type") or self.config.get("search_type") or "hybrid" + + def _retry_config(self) -> Dict[str, Any]: + """The evidence-sufficiency retry block (roadmap item 2.1). + + Merged across both container spellings so a runtime override written by + the API under ``retrievers.retry`` beats the profile's + ``retrieval.retry``, matching how ``_latechunk_config`` behaves. + """ + merged: Dict[str, Any] = {} + for container_key in ("retrieval", "retrievers"): + block = (self.config.get(container_key) or {}).get("retry") + if isinstance(block, dict): + merged.update(block) + return merged + + def _merged_block(self, key: str) -> Dict[str, Any]: + """A config block merged across the ``retrieval`` / ``retrievers`` spellings. + + Same layering as ``_retry_config``: a runtime override written by the API + under ``retrievers.<key>`` beats the profile's ``retrieval.<key>``. + """ + merged: Dict[str, Any] = {} + for container_key in ("retrieval", "retrievers"): + block = (self.config.get(container_key) or {}).get(key) + if isinstance(block, dict): + merged.update(block) + return merged + + def _crossref_hop_config(self) -> Dict[str, Any]: + """``retrieval.crossref_hop`` (roadmap item 4.2). Default OFF.""" + return self._merged_block("crossref_hop") + + def _overview_prefilter_config(self) -> Dict[str, Any]: + """``retrieval.overview_prefilter`` (roadmap item 4.3). Default OFF.""" + return self._merged_block("overview_prefilter") + + def _latechunk_config(self) -> Dict[str, Any]: + """Merge the late-chunk block across both container and key spellings. + + The profile declares it under ``retrieval.late_chunking`` while the API + toggles it at runtime under ``retrievers.latechunk``; later writes win. + """ + merged: Dict[str, Any] = {} + for container_key in ("retrieval", "retrievers"): + container = self.config.get(container_key) or {} + for key in ("late_chunking", "latechunk"): + block = container.get(key) + if isinstance(block, dict): + merged.update(block) + return merged + + @staticmethod + def _latechunk_table_name(latechunk_cfg: Dict[str, Any], base_table: str) -> Optional[str]: + explicit = latechunk_cfg.get("lancedb_table_name") + if explicit: + return explicit + # "_lc" is the suffix IndexingPipeline writes when none is configured. + suffix = latechunk_cfg.get("table_suffix", "_lc") + return f"{base_table}{suffix}" if suffix and base_table else None + def _get_db_manager(self): if self.db_manager is None: # Accept either "db_path" (preferred) or legacy "lancedb_uri" @@ -72,66 +192,79 @@ def _get_db_manager(self): self.db_manager = LanceDBManager(db_path=db_path) return self.db_manager + def _query_instruction(self, model_name: str) -> str: + """The query-side instruction prefix for this pipeline's embedder. + + Resolution order, most explicit first: + + 1. ``config["embedding_instruction"]`` โ€” set it to ``""`` to switch the + prefix off for a model whose family would otherwise get one. + 2. ``EMBEDDING_INSTRUCTION`` env var โ€” same semantics, for A/B runs. + 3. The model family's official retrieval instruction (Qwen3-Embedding + and harrier-oss-v1), or ``""`` for everything else. + + This is the QUERY side only. Documents are embedded by + ``IndexingPipeline``, which calls ``select_embedder`` without an + instruction, so an index built before this existed remains valid. + """ + configured = self.config.get("embedding_instruction") + if configured is not None: + return configured + env = os.getenv("EMBEDDING_INSTRUCTION") + if env is not None: + return env + return default_query_instruction(model_name) + def _get_text_embedder(self): if self.text_embedder is None: - from rag_system.indexing.representations import select_embedder + model_name = self.config.get("embedding_model_name") + if not model_name: + raise ValueError( + "Config must contain 'embedding_model_name'. Falling back to a hard-coded " + "default here would silently produce vectors whose dimensionality does not " + "match the index." + ) + instruction = self._query_instruction(model_name) + if instruction: + print(f"๐Ÿ”ง Query-side embedding instruction active: '{instruction}'") self.text_embedder = select_embedder( - self.config.get("embedding_model_name", "BAAI/bge-small-en-v1.5"), + model_name, self.ollama_config.get("host") if isinstance(self.ollama_config, dict) else None, + query_instruction=instruction, ) return self.text_embedder def _get_dense_retriever(self): - """Ensure a dense MultiVectorRetriever is always available unless explicitly disabled.""" + """Ensure a MultiVectorRetriever is always available unless explicitly disabled. + + A failed construction is cached in ``_dense_retriever_error``: the first + failure logs and returns None (the pipeline proceeds without dense, as + before), while every later call re-raises a concise error instead of + retrying the failing constructor and reprinting the error per query. + ``update_embedding_model`` clears the cache so a model switch can retry. + """ if self.dense_retriever is None: # If the config explicitly sets dense.enabled to False, respect it - if self.retriever_configs.get("dense", {}).get("enabled", True) is False: + if self._retriever_config("dense").get("enabled", True) is False: return None + if self._dense_retriever_error is not None: + raise RuntimeError( + "Dense retriever initialisation previously failed " + f"({self._dense_retriever_error}); not retrying the constructor " + "on every query. Fix the cause or switch embedding model." + ) + try: - db_manager = self._get_db_manager() - text_embedder = self._get_text_embedder() - fusion_cfg = self.config.get("fusion", {}) self.dense_retriever = MultiVectorRetriever( - db_manager, - text_embedder, - vision_model=None, - fusion_config=fusion_cfg, + self._get_db_manager(), + self._get_text_embedder(), ) except Exception as e: print(f"โŒ Failed to initialise dense retriever: {e}") - self.dense_retriever = None + self._dense_retriever_error = e return self.dense_retriever - def _get_bm25_retriever(self): - if self.bm25_retriever is None and self.retriever_configs.get("bm25", {}).get("enabled"): - try: - print(f"๐Ÿ”ง Lazily initializing BM25 retriever...") - self.bm25_retriever = BM25Retriever( - index_path=self.storage_config["bm25_path"], - index_name=self.retriever_configs["bm25"]["index_name"] - ) - print("โœ… BM25 retriever initialized successfully") - except Exception as e: - print(f"โŒ Failed to initialize BM25 retriever on demand: {e}") - # Keep it None so we don't try again - return self.bm25_retriever - - def _get_graph_retriever(self): - if self._graph_retriever is None and self.retriever_configs.get("graph", {}).get("enabled"): - self._graph_retriever = GraphRetriever(graph_path=self.storage_config["graph_path"]) - return self._graph_retriever - - def _get_reranker(self): - """Initializes the reranker for hybrid search score fusion.""" - reranker_config = self.config.get("reranker", {}) - # This is for the LanceDB internal reranker, not the AI one. - if self.reranker is None and reranker_config.get("type") == "linear_combination": - rerank_weight = reranker_config.get("weight", 0.5) - self.reranker = lancedb.rerankers.LinearCombinationReranker(weight=rerank_weight) - print(f"โœ… Initialized LinearCombinationReranker with weight {rerank_weight}") - return self.reranker - def _get_ai_reranker(self): """Initializes a dedicated AI-based reranker.""" reranker_config = self.config.get("reranker", {}) @@ -142,22 +275,38 @@ def _get_ai_reranker(self): with _ai_reranker_init_lock: # Another thread may have completed init while we waited if self.ai_reranker is None: + model_name = reranker_config.get("model_name") + if not model_name: + print("โš ๏ธ Reranking is enabled but 'reranker.model_name' is not configured; skipping reranking.") + return None try: - model_name = reranker_config.get("model_name") - strategy = reranker_config.get("strategy", "qwen") + strategy = reranker_config.get("strategy", "rerankers-lib") + model_type = reranker_config.get("model_type", "cross-encoder") - if strategy == "rerankers-lib": - print(f"๐Ÿ”ง Initialising Answer.AI ColBERT reranker ({model_name}) via rerankers libโ€ฆ") + # Qwen3-Reranker is a causal-LM yes/no-logit scorer, not a + # SequenceClassification model. The rerankers lib silently + # loads it with a randomly-initialised score head, so route + # the whole family to our own scorer โ€” either by explicit + # `reranker.model_type: "qwen3"` or by model name. + if model_type == "qwen3" or is_qwen3_reranker(model_name): + print(f"๐Ÿ”ง Initialising Qwen3 yes/no-logit reranker ({model_name})โ€ฆ") + self.ai_reranker = QwenRerankerScorer(model_name=model_name) + elif strategy == "maxsim": + print(f"๐Ÿ”ง Initialising MaxSim late-interaction rescorer ({model_name}) via sidecarโ€ฆ") + from rag_system.rerankers.reranker import MaxSimRerankerScorer + self.ai_reranker = MaxSimRerankerScorer() + elif strategy == "rerankers-lib": + print(f"๐Ÿ”ง Initialising {model_type} reranker ({model_name}) via rerankers libโ€ฆ") from rerankers import Reranker - self.ai_reranker = Reranker(model_name, model_type="colbert") + self.ai_reranker = Reranker(model_name, model_type=model_type) else: - print(f"๐Ÿ”ง Lazily initializing Qwen reranker ({model_name})โ€ฆ") - self.ai_reranker = QwenReranker(model_name=model_name) + print(f"๐Ÿ”ง Lazily initializing local cross-encoder reranker ({model_name})โ€ฆ") + self.ai_reranker = CrossEncoderReranker(model_name=model_name) print("โœ… AI reranker initialized successfully.") except Exception as e: # Leave as None so the pipeline can proceed without reranking - print(f"โŒ Failed to initialize AI reranker: {e}") + print(f"โš ๏ธ Could not load reranker '{model_name}' ({e}). Continuing without reranking.") return self.ai_reranker def _get_sentence_pruner(self): @@ -167,9 +316,130 @@ def _get_sentence_pruner(self): self._sentence_pruner = SentencePruner() return self._sentence_pruner - def _get_surrounding_chunks_lancedb(self, chunk: Dict[str, Any], window_size: int) -> List[Dict[str, Any]]: + # ------------------------------------------------------------------ + # Evidence-sufficiency retry (roadmap item 2.1) + # ------------------------------------------------------------------ + + # Candidates from this rank onwards are treated as "background": chunks the + # query pulled in because they are documents in this corpus, not because + # they answer it. Rank 6 of 20 leaves the plausible answers out of the + # background estimate while still averaging over enough rows to be stable. + _EVIDENCE_BACKGROUND_FROM = 5 + + @classmethod + def _dense_evidence_score(cls, docs: List[Dict[str, Any]]) -> Optional[float]: + """A 0โ€“1 "did we actually find something" score from the dense leg. + + The naive choice โ€” the raw top cosine similarity โ€” was **measured and + rejected**: on the gold set it is *anti*-correlated with success, + because absolute similarity mostly encodes how close the query's + phrasing sits to the corpus's register, not whether the answer-bearing + chunk was found. The three `mixed` first-stage misses all scored a + *higher* top cosine than the median successful query. + + What does carry signal is **contrast**: how far the best candidate + stands above the background of everything else the query pulled in. + + score = (cos_top โˆ’ cos_background) / (1 โˆ’ cos_background) + + where ``cos_background`` is the mean cosine of the candidates from rank + ``_EVIDENCE_BACKGROUND_FROM`` down. The denominator rescales against the + headroom that is actually reachable for this query, keeping the result + in 0โ€“1 and comparable across queries whose background level differs. + + Returns ``None`` when the dense leg did not run (``fts_only``) or the + table predates cosine normalization, in which case the caller must not + retry โ€” the number would not mean anything. Requires L2-normalized + vectors (v4+ tables), where LanceDB's squared-L2 ``_distance`` maps to + cosine as ``cos = 1 โˆ’ d/2``. + """ + sims = [] + for doc in docs: + distance = doc.get("_distance") + if distance is None: + continue + try: + sims.append(1.0 - float(distance) / 2.0) + except (TypeError, ValueError): + continue + if len(sims) < 2: + return None + sims.sort(reverse=True) + tail = sims[cls._EVIDENCE_BACKGROUND_FROM:] or sims[1:] + background = sum(tail) / len(tail) + headroom = 1.0 - background + if headroom <= 1e-6: + return None + return max(0.0, min(1.0, (sims[0] - background) / headroom)) + + @staticmethod + def _rerank_evidence_score(docs: List[Dict[str, Any]]) -> Optional[float]: + """Top reranker score, when the reranker produces a calibrated 0โ€“1 one. + + ``QwenRerankerScorer`` returns P("yes") per candidate, which is directly + interpretable. Other backends return arbitrary logits, so anything + outside 0โ€“1 is rejected rather than silently compared to a probability + threshold. + """ + scores = [d.get("rerank_score") for d in docs if d.get("rerank_score") is not None] + if not scores: + return None + try: + top = float(max(scores)) + except (TypeError, ValueError): + return None + if not (0.0 <= top <= 1.0): + return None + return top + + def _reformulate_query(self, query: str) -> Optional[str]: + """One rewrite of a query whose first pass found weak evidence. + + Runs on the enrichment (utility) model, not the generation model, and + asks for JSON so the "thinking" preamble small models emit cannot leak + into the rewritten query. + """ + model = (self.ollama_config.get("enrichment_model") + or self.ollama_config.get("generation_model")) + if not model: + return None + prompt = ( + "A document search for the question below returned weak matches.\n" + "Rewrite it once as a single self-contained search query that uses the " + "concrete nouns, technical terms and synonyms a document would actually " + "use, instead of the asker's phrasing. Keep every entity, number and " + "constraint from the original. Do not answer the question.\n\n" + f'Question: "{query}"\n\n' + 'Respond with JSON: {"query": "<rewritten query>"}' + ) + try: + resp = self.ollama_client.generate_completion( + model=model, prompt=prompt, format="json", + options={"temperature": 0}) # greedy decode: the retry must be reproducible + data = json.loads(resp.get("response", "{}")) + except Exception as e: + print(f"โš ๏ธ Retry reformulation failed ({e}); keeping the original query.") + return None + rewritten = (data.get("query") or "").strip() if isinstance(data, dict) else "" + if not rewritten or rewritten.lower() == query.strip().lower(): + return None + return rewritten + + def _get_surrounding_chunks_lancedb(self, chunk: Dict[str, Any], window_size: int, + *, table_name: Optional[str] = None, + active_filter: Optional[CompiledFilter] = None + ) -> List[Dict[str, Any]]: """ Retrieves a window of chunks around a central chunk using LanceDB. + + *table_name* overrides the configured text table; ``run()`` passes the + table it resolved so a per-call table no longer has to be smuggled + through shared config. The default keeps direct callers on the config + table. *active_filter* is the caller's compiled metadata filter, + captured in the submitting thread: this method also runs on + ThreadPoolExecutor workers, which do not inherit the caller's + thread-local filter scope. None means "read the thread-local scope", + which is correct for direct (same-thread) callers. """ db_manager = self._get_db_manager() if not db_manager: @@ -183,7 +453,16 @@ def _get_surrounding_chunks_lancedb(self, chunk: Dict[str, Any], window_size: in if document_id is None or chunk_index is None or chunk_index == -1: return [chunk] - table_name = self.config["storage"]["text_table_name"] + # document_id derives from upload filenames and is interpolated into a + # SQL literal below. Refuse ids that could terminate the literal rather + # than escaping them (the filters.py / document_fetch.py standard) and + # fail closed to the single chunk, exactly like the query-failure path. + if "'" in document_id or "\\" in document_id: + print(f"โš ๏ธ Refusing context expansion for document id with quoting " + f"characters: {document_id!r}") + return [chunk] + + table_name = table_name or self.config["storage"]["text_table_name"] try: tbl = db_manager.get_table(table_name) except Exception: @@ -196,7 +475,13 @@ def _get_surrounding_chunks_lancedb(self, chunk: Dict[str, Any], window_size: in # Construct the SQL filter for an efficient metadata-based search sql_filter = f"document_id = '{document_id}' AND chunk_index >= {start_index} AND chunk_index <= {end_index}" - + # A caller's metadata filter (item 4.4) also bounds context expansion: + # otherwise a `chunk_index <= 0` filter would still pull chunk 1 back in + # as a neighbour, and the guarantee "nothing that fails the filter + # reaches synthesis" would be false. + sql_filter = combine(sql_filter, active_filter if active_filter is not None + else self.active_filter()) + try: # Execute a filter-only search, which is very fast on indexed metadata results = tbl.search().where(sql_filter).to_list() @@ -216,23 +501,78 @@ def _get_surrounding_chunks_lancedb(self, chunk: Dict[str, Any], window_size: in # If the query fails for any reason, fall back to the single chunk return [chunk] + def _budget_synthesis_context(self, final_docs): + """Fit the synthesis input to an explicit token budget (top-rank first). + + Measured on the rfc corpus before this existed: dual-leg retrieval, + enrichment and ยฑ1 sibling merging compounded 20 nominal chunks into + ~94k tokens per synthesis call, which the serving layer then silently + front-truncated to its slot window โ€” deleting the TOP-ranked evidence + and leaving the tail. Packing to a budget hands the model exactly the + best-ranked content that fits, and what it sees is what we cite. + + Two mechanisms, both rank-order-preserving: + * overlap suppression โ€” a sibling-merged doc covers chunk_index ยฑ1; + a later doc whose whole coverage is already included adds nothing + and is skipped; + * token budget (``retrieval.synthesis_context_tokens``, default + 12000, chars/3.5 estimate) โ€” packing stops when adding the next + doc would exceed it; at least one doc is always kept. + """ + if not final_docs: + return final_docs + try: + budget = int(self._merged_block("synthesis_context") .get("tokens", 0)) or \ + int((self.config.get("retrieval") or {}).get("synthesis_context_tokens", 0)) or 12000 + except Exception: + budget = 12000 + + kept, covered, used = [], set(), 0 + skipped_overlap = skipped_budget = 0 + for doc in final_docs: + doc_id = doc.get("document_id") + cidx = doc.get("chunk_index") + merged = bool((doc.get("metadata") or {}).get("latechunk_merged")) + if doc_id is not None and cidx is not None and cidx != -1: + span = {(doc_id, cidx + off) for off in ((-1, 0, 1) if merged else (0,))} + else: + span = set() + if span and span <= covered: + skipped_overlap += 1 + continue + tokens = int(len(doc.get("text") or "") / 3.5) + 1 + if kept and used + tokens > budget: + skipped_budget += 1 + continue + kept.append(doc) + covered |= span + used += tokens + if skipped_overlap or skipped_budget: + print(f"โœ‚๏ธ Synthesis context budget: kept {len(kept)}/{len(final_docs)} docs " + f"(~{used} tokens, budget {budget}; {skipped_overlap} overlap-dup, " + f"{skipped_budget} over-budget dropped)") + return kept + def _synthesize_final_answer(self, query: str, facts: str, *, event_callback=None) -> str: """Uses a text LLM to synthesize a final answer from extracted facts.""" + # Arm-C prompt from the synthesis-grounding A/B + # (eval/decisions/synthesis-grounding-ab-2026-08-13.md): deletes the + # old "General knowledge" escape hatch that produced fabricated + # citations on unseen corpora, and forbids quotes/section numbers/ + # document names not present in the snippets. prompt = f""" -You are an AI assistant specialised in answering questions from retrieved context. +You are answering strictly from the retrieved snippets below. -Context you receive -โ€ข VERIFIED FACTS โ€“ text snippets retrieved from the user's documents. Some may be irrelevant noise. -โ€ข ORIGINAL QUESTION โ€“ the user's actual query. - -Instructions -1. Evaluate each snippet for relevance to the ORIGINAL QUESTION; ignore those that do not help answer it. -2. Synthesise an answer **using only information from the relevant snippets**. -3. If snippets contradict one another, mention the contradiction explicitly. -4. If the snippets do not contain the needed information, reply exactly with: - "I could not find that information in the provided documents." -5. Provide a thorough, well-structured answer. Use paragraphs or bullet points where helpful, and include any relevant numbers/names exactly as they appear. There is **no strict sentence limit**, but aim for clarity over brevity. -6. Do **not** introduce external knowledge unless step 4 applies; in that case you may add a clearly-labelled "General knowledge" sentence after the required statement. +Hard rules โ€” these override anything you believe you know: +1. Use ONLY information stated in the snippets. Your own knowledge of the topic, however confident, must not appear in the answer. +2. If the snippets disagree with what you remember, the snippets are correct. +3. Copy every number, identifier, code and quoted phrase character-for-character from a snippet. Never write a quotation, section number or document name that does not appear in the snippets. +4. If the snippets do not contain the needed information, reply exactly: + "I could not find that information in the provided documents." + Do not add a general-knowledge answer after it. +5. If snippets contradict one another, state the contradiction explicitly. +6. Be thorough and well-structured, but stay within the snippets; include relevant numbers and names exactly as they appear. +7. Each snippet starts with a "[Source document: โ€ฆ]" line naming the file it came from. When the question asks where something is defined, or attributing a fact matters, name that source document. Output format Answer: @@ -244,11 +584,16 @@ def _synthesize_final_answer(self, query: str, facts: str, *, event_callback=Non ORIGINAL QUESTION: "{query}" """ - # Stream the answer token-by-token so the caller can forward them as SSE + # Stream the answer token-by-token so the caller can forward them as SSE. + # Thinking must be OFF here: with it on, the model spends its window on + # chain-of-thought that never enters `response` and can return an empty + # answer (measured: prompt 9351 + thinking 7033 = window exactly, "" out). answer_parts: list[str] = [] for tok in self.ollama_client.stream_completion( model=self.ollama_config["generation_model"], prompt=prompt, + enable_thinking=False, + options={"temperature": 0}, # greedy decode: measured fewer prior-driven drifts, zero judge splits ): answer_parts.append(tok) if event_callback: @@ -256,60 +601,346 @@ def _synthesize_final_answer(self, query: str, facts: str, *, event_callback=Non return "".join(answer_parts) - def run(self, query: str, table_name: str = None, window_size_override: Optional[int] = None, event_callback=None) -> Dict[str, Any]: - start_time = time.time() - retrieval_k = self.config.get("retrieval_k", 10) + # ------------------------------------------------------------------ + # Document-scoped search (shared by the cross-reference hop, item 4.2, + # and the overview prefilter's "restrict" mode, item 4.3) + # ------------------------------------------------------------------ - logger = logging.getLogger(__name__) - logger.debug("--- Running Hybrid Search for query '%s' (table=%s) ---", query, table_name or self.storage_config.get("text_table_name")) - - # If a custom table_name is provided, propagate it to storage config so helper methods use it - if table_name: - self.storage_config["text_table_name"] = table_name + def _embed_query(self, query: str): + """The query vector, reusing the retriever's LRU cache when it exists.""" + retriever = self._get_dense_retriever() + cached = getattr(retriever, "_embed_single", None) if retriever else None + if cached is not None: + return cached(query) + return self._get_text_embedder().create_embeddings([query])[0] - if event_callback: - event_callback("retrieval_started", {}) - # Unified retrieval using the refactored MultiVectorRetriever + def _table_normalizes(self, tbl, table_name: str) -> bool: + """Whether *tbl* holds L2-normalized vectors; also re-checks the embedder.""" + marker = read_table_marker(tbl, getattr(self._get_db_manager(), "db_path", None), + table_name) + if marker is None: + return False + configured = self.config.get("embedding_model_name") + if configured: + assert_embedder_matches(table_name, marker, configured) + return bool(marker["normalized"]) + + @staticmethod + def _doc_id_filter(doc_ids: List[str]) -> str: + quoted = ", ".join("'" + str(d).replace("'", "''") + "'" for d in doc_ids) + return f"document_id IN ({quoted})" + + @staticmethod + def _row_to_doc(row: Dict[str, Any], score: float) -> Dict[str, Any]: + """A LanceDB row in the shape ``MultiVectorRetriever.retrieve`` returns.""" + raw_metadata = row.get("metadata") + if isinstance(raw_metadata, dict): + metadata = dict(raw_metadata) + else: + try: + metadata = json.loads(raw_metadata or "{}") + except (TypeError, ValueError): + metadata = {} + metadata.setdefault("document_id", row.get("document_id")) + metadata.setdefault("chunk_index", row.get("chunk_index")) + doc = { + "chunk_id": row.get("chunk_id"), + "text": metadata.get("original_text") or row.get("text") or "", + "score": score, + "document_id": row.get("document_id"), + "chunk_index": row.get("chunk_index"), + "metadata": metadata, + } + distance = row.get("_distance") + if distance is not None: + try: + doc["_distance"] = float(distance) + except (TypeError, ValueError): + pass + return doc + + def _search_within_documents(self, query: str, table_name: str, doc_ids: List[str], + k: int, mode: str = "vector_only") -> Optional[List[Dict[str, Any]]]: + """Retrieve up to *k* chunks, restricted to *doc_ids* with a LanceDB filter. + + ``MultiVectorRetriever.retrieve`` has no filter parameter and belongs to + another module, so the document-scoped variant lives here. It mirrors the + retriever exactly โ€” same prefiltered legs, same RRF fusion at the same + ``_RRF_K``, same output shape โ€” so a chunk pulled by a hop is + indistinguishable downstream from one pulled by the first stage. + + Returns ``None`` โ€” not an empty list โ€” when the search could not run at + all (table cannot be opened, or every leg the mode asked for failed), + so the caller can tell "the restriction did not run" apart from "the + restriction matched nothing". Only the latter may be widened into an + unrestricted search. + """ + if not doc_ids or k < 1: + return [] + try: + tbl = self._get_db_manager().get_table(table_name) + normalize = self._table_normalizes(tbl, table_name) + except EmbedderMismatchError: + raise + except Exception as e: + print(f"โš ๏ธ Document-scoped search: cannot open table '{table_name}': {e}") + return None + + # A caller's metadata filter (item 4.4) narrows every internally-scoped + # search too. The cross-reference hop must not be a way to reach a + # document the caller filtered out. + where = combine(self._doc_id_filter(doc_ids), self.active_filter()) + mode = (mode or "vector_only").lower() + fts_rows: List[Dict[str, Any]] = [] + vec_rows: List[Dict[str, Any]] = [] + leg_ran = False + leg_failed = False + + if mode != "fts_only": + try: + vector = self._embed_query(query) + if normalize: + vector = l2_normalize(vector) + vec_rows = (tbl.search(vector).where(where, prefilter=True) + .limit(k).to_pandas().to_dict("records")) + leg_ran = True + except Exception as e: + leg_failed = True + print(f"โš ๏ธ Document-scoped vector search failed: {e}") + if mode != "vector_only": + try: + # Same quote-stripping as MultiVectorRetriever._run_fts: + # LanceDB's FTS parser reads double quotes as phrase syntax and + # raises on quoted decomposer output, which would kill this leg. + fts_query = query.replace('"', " ").strip() or query + if len(fts_query.split()) == 1: + fts_query = f"{fts_query}* OR {fts_query}~" + fts_rows = (tbl.search(query=fts_query, query_type="fts") + .where(where, prefilter=True) + .limit(k).to_pandas().to_dict("records")) + leg_ran = True + except Exception as e: + leg_failed = True + print(f"โš ๏ธ Document-scoped full-text search failed: {e}") + + if leg_failed and not leg_ran: + # Every leg the mode asked for failed: the restriction never ran. + return None + + fused: Dict[Any, Dict[str, Any]] = {} + for rows in (fts_rows, vec_rows): + for rank, row in enumerate(rows, start=1): + key = row.get("chunk_id") or row.get("text") + entry = fused.setdefault(key, {"row": row, "rrf": 0.0}) + entry["rrf"] += 1.0 / (_RRF_K + rank) + ordered = sorted(fused.values(), key=lambda e: e["rrf"], reverse=True)[:k] + return [self._row_to_doc(e["row"], e["rrf"]) for e in ordered] + + # ------------------------------------------------------------------ + # Overview prefilter (roadmap item 4.3) + # ------------------------------------------------------------------ + + def _overview_vectors(self) -> Optional[Dict[str, Any]]: + """The embedded-overview sidecar for this index, or None (logged once). + + Path resolution, most explicit first: + + 1. ``retrieval.overview_prefilter.vectors_path`` + 2. ``config["overview_path"]`` with its ``.jsonl`` swapped for + ``.vectors.npz`` โ€” this is what ``api_server`` already sets per + session, so the HTTP path needs no extra plumbing. + 3. ``index_store/overviews/<index_id>.vectors.npz``. + + Nothing is guessed beyond that: silently prefiltering a query against + *some other index's* overviews would be worse than not prefiltering. + """ + if self._overview_vectors_loaded: + return self._overview_vectors_data + self._overview_vectors_loaded = True + + cfg = self._overview_prefilter_config() + path = cfg.get("vectors_path") + if not path: + overview_path = (self.config.get("overview_path") + or (self.config.get("overview") or {}).get("path")) + if overview_path: + path = overview_vectors_path(overview_path) + if not path: + index_id = cfg.get("index_id") or self.config.get("index_id") + if index_id: + path = os.path.join("index_store", "overviews", f"{index_id}.vectors.npz") + if not path: + print("โ„น๏ธ Overview prefilter is on but no overview path is configured " + "(set retrieval.overview_prefilter.vectors_path or overview_path); " + "continuing without it.") + return None + + data = load_overview_vectors(path) + if data is None: + print(f"โ„น๏ธ Overview prefilter is on but no embedded overviews were found at " + f"{path}; continuing without it. Re-index to build the sidecar.") + return None + + recorded = (data.get("meta") or {}).get("embedding_model") + configured = self.config.get("embedding_model_name") + if recorded and configured and recorded != configured: + print(f"โ„น๏ธ Overview prefilter disabled: the sidecar at {path} was written by " + f"'{recorded}' but this pipeline uses '{configured}'.") + return None + + print(f"๐Ÿงญ Overview prefilter: {len(data['doc_ids'])} document overview(s) loaded " + f"from {path}.") + self._overview_vectors_data = data + return data + + def _overview_prefilter_documents(self, query: str) -> Optional[List[str]]: + """The top-N document ids by query-vs-overview similarity, or None.""" + cfg = self._overview_prefilter_config() + if not cfg.get("enabled"): + return None + data = self._overview_vectors() + if data is None: + return None + top_n = int(cfg.get("top_documents", 5) or 0) + if top_n < 1: + return None + try: + vector = l2_normalize(self._embed_query(query)) + scores = np.asarray(data["vectors"], dtype="float32") @ np.asarray(vector, + dtype="float32") + except Exception as e: + print(f"โš ๏ธ Overview prefilter scoring failed ({e}); continuing without it.") + return None + order = np.argsort(-scores)[:top_n] + selected = [data["doc_ids"][int(i)] for i in order] + print(f"๐Ÿงญ Overview prefilter selected {len(selected)} document(s): " + f"{', '.join(selected)}") + return selected + + @staticmethod + def _apply_overview_boost(docs: List[Dict[str, Any]], + prefiltered: List[str]) -> List[Dict[str, Any]]: + """Fuse the candidate ordering with the document-overview ordering by RRF. + + A rank bonus rather than a score bonus, and RRF rather than a weighted + sum, for the same reason the retriever fuses its two legs that way: the + two orderings are not on a common scale and there is no validation split + here to tune a weight against (design_rationale ยง4). + """ + doc_rank = {doc_id: rank for rank, doc_id in enumerate(prefiltered)} + scored = [] + for position, doc in enumerate(docs): + fused = 1.0 / (_RRF_K + position + 1) + rank = doc_rank.get(doc.get("document_id")) + if rank is not None: + fused += 1.0 / (_RRF_K + rank + 1) + doc["overview_prefilter_rank"] = rank + scored.append((fused, position, doc)) + scored.sort(key=lambda item: (-item[0], item[1])) + return [doc for _, _, doc in scored] + + def _first_stage(self, query: str, base_table: str, retrieval_k: int, retrieval_mode: str, + event_callback=None) -> List[Dict[str, Any]]: + """Hybrid/vector/FTS retrieval plus the optional late-chunk table and merge. + + Split out of ``run()`` so the evidence-sufficiency retry can call it a + second time with a reformulated query without duplicating any of it. + + The caller's metadata filter (item 4.4) is read from the thread-local + scope rather than passed in, so the retry re-applies it automatically + and no existing call site changes. + """ + start_time = time.time() + logger = logging.getLogger(__name__) dense_retriever = self._get_dense_retriever() - # Get the LanceDB reranker for initial score fusion - lancedb_reranker = self._get_reranker() - + active_filter = self.active_filter() + filter_where = active_filter.where if active_filter is not None else None + + # Overview prefilter (roadmap item 4.3). Computed here rather than in + # retrieve_candidates so the evidence-sufficiency retry re-scores the + # documents against its reformulated query too. + prefilter_cfg = self._overview_prefilter_config() + prefilter_mode = (prefilter_cfg.get("mode") or "boost").lower() + prefilter_docs = self._overview_prefilter_documents(query) + restrict_docs = prefilter_docs if (prefilter_docs and prefilter_mode == "restrict") else None + retrieved_docs = [] - if dense_retriever: + restrict_failed = False + if restrict_docs: + restricted = self._search_within_documents( + query, base_table, restrict_docs, retrieval_k, retrieval_mode) + if restricted is None: + # The restricted search never ran; do NOT widen that into an + # unrestricted retrieval. "Search failed" and "matched nothing" + # are different answers and only the second is safe to widen. + print("โš ๏ธ Overview prefilter (restrict) search failed; not falling " + "back to unrestricted retrieval for this query.") + restrict_failed = True + elif restricted: + retrieved_docs = restricted + else: + print("โš ๏ธ Overview prefilter (restrict) matched no chunks; falling back " + "to unrestricted retrieval for this query.") + restrict_docs = None + if not retrieved_docs and not restrict_failed and dense_retriever: retrieved_docs = dense_retriever.retrieve( text_query=query, - table_name=table_name or self.storage_config["text_table_name"], + table_name=base_table, k=retrieval_k, - reranker=lancedb_reranker # Pass the reranker to enable hybrid search + search_type=retrieval_mode, + where=filter_where, ) # --------------------------------------------------------------- # Late-Chunk retrieval (optional) # --------------------------------------------------------------- - if self.retriever_configs.get("latechunk", {}).get("enabled"): - lc_table = self.retriever_configs["latechunk"].get("lancedb_table_name") + latechunk_cfg = self._latechunk_config() + if dense_retriever and latechunk_cfg.get("enabled"): + lc_table = self._latechunk_table_name(latechunk_cfg, base_table) if lc_table: try: - lc_docs = dense_retriever.retrieve( - text_query=query, - table_name=lc_table, - k=retrieval_k, - reranker=lancedb_reranker, - ) + if restrict_docs: + lc_docs = self._search_within_documents( + query, lc_table, restrict_docs, retrieval_k, + retrieval_mode) or [] + else: + lc_docs = dense_retriever.retrieve( + text_query=query, + table_name=lc_table, + k=retrieval_k, + search_type=retrieval_mode, + where=filter_where, + ) retrieved_docs.extend(lc_docs) except Exception as e: print(f"โš ๏ธ Late-chunk retrieval failed: {e}") + # Dedupe across the two legs (improvement_plan 1.5): the base and + # late-chunk tables index the same passages, so the same + # (document, chunk_index) can arrive twice and occupy two of the + # candidate slots. First (higher-ranked) instance wins. + if retrieved_docs: + seen_keys = set() + deduped = [] + for doc in retrieved_docs: + key = (doc.get("document_id"), doc.get("chunk_index")) + if key in seen_keys and key != (None, None): + continue + seen_keys.add(key) + deduped.append(doc) + if len(deduped) != len(retrieved_docs): + print(f"๐Ÿ” Cross-leg dedupe: {len(retrieved_docs)} โ†’ {len(deduped)} candidates.") + retrieved_docs = deduped + if event_callback: event_callback("retrieval_done", {"count": len(retrieved_docs)}) - - retrieval_time = time.time() - start_time - logger.debug("Retrieved %s chunks in %.2fs", len(retrieved_docs), retrieval_time) + + logger.debug("Retrieved %s chunks in %.2fs", len(retrieved_docs), time.time() - start_time) # ----------------------------------------------------------- # LATE-CHUNK MERGING (merge ยฑ1 sub-vector into central hit) # ----------------------------------------------------------- - if self.retriever_configs.get("latechunk", {}).get("enabled") and retrieved_docs: + if latechunk_cfg.get("enabled") and retrieved_docs: merged_count = 0 for doc in retrieved_docs: try: @@ -322,12 +953,19 @@ def run(self, query: str, table_name: str = None, window_size_override: Optional if doc_id is None or cidx is None or cidx == -1: continue # Fetch neighbouring late-chunks inside same document (ยฑ1) - siblings = self._get_surrounding_chunks_lancedb(doc, window_size=1) + siblings = self._get_surrounding_chunks_lancedb( + doc, window_size=1, table_name=base_table) # Keep only same document_id and ordered by chunk_index siblings = [s for s in siblings if s.get("document_id") == doc_id] siblings.sort(key=lambda s: s.get("chunk_index", 0)) merged_text = " \n".join(s.get("text", "") for s in siblings) if merged_text: + # The retrieved chunk is what matched the query; the + # siblings are context for synthesis. Keep the core + # text so the reranker can score what actually matched + # (the merged text buries it mid-string, past the + # scorer's truncation point). + meta["core_text"] = doc.get("text") doc["text"] = merged_text meta["latechunk_merged"] = True merged_count += 1 @@ -336,110 +974,578 @@ def run(self, query: str, table_name: str = None, window_size_override: Optional if merged_count: print(f"๐Ÿช„ Late-chunk merging applied to {merged_count} retrieved chunks.") - # --- AI Reranking Step --- - ai_reranker = self._get_ai_reranker() - if ai_reranker and retrieved_docs: - if event_callback: - event_callback("rerank_started", {"count": len(retrieved_docs)}) - print(f"\n--- Reranking top {len(retrieved_docs)} docs with AI model... ---") - start_rerank_time = time.time() + if prefilter_docs and prefilter_mode == "boost" and retrieved_docs: + before = [d.get("chunk_id") for d in retrieved_docs[:3]] + retrieved_docs = self._apply_overview_boost(retrieved_docs, prefilter_docs) + if [d.get("chunk_id") for d in retrieved_docs[:3]] != before: + print("๐Ÿงญ Overview prefilter (boost) reordered the candidate list.") - rerank_cfg = self.config.get("reranker", {}) - top_k_cfg = rerank_cfg.get("top_k") - top_percent = rerank_cfg.get("top_percent") # value in range 0โ€“1 + return retrieved_docs - if top_percent is not None: - try: - pct = float(top_percent) - assert 0 < pct <= 1 - top_k = max(1, int(len(retrieved_docs) * pct)) - except Exception: - print("โš ๏ธ Invalid top_percent value; falling back to top_k") - top_k = top_k_cfg or len(retrieved_docs) - else: - top_k = top_k_cfg or len(retrieved_docs) + # ------------------------------------------------------------------ + def _pooled_first_stage(self, sub_queries: List[str], base_table: str, + retrieval_k: int, retrieval_mode: str, + event_callback=None) -> List[Dict[str, Any]]: + """Run the first stage once per sub-query and pool the results. + + Decomposed queries used to run one full pipeline (retrieval + rerank + + synthesis) per sub-query and compose the sub-ANSWERS. This pools the + per-sub-query *retrievals* instead, so one rerank/selection pass and + one synthesis over the union context replace N of each โ€” and the + composer (where multi-hop facts were measurably lost, arm E) drops + out entirely. + + Each pooled doc is tagged with the indices of the sub-queries that + retrieved it (``_source_sqs``, consumed and removed by + ``_rerank_stage``), so reranking can score every candidate against + exactly the sub-queries that found it โ€” union-of-max, at the same + pair cost as the old per-sub-query passes. Pool order interleaves the + per-sub-query rankings round-robin so no sub-question monopolizes the + front of the list if reranking is off. + """ + per_sq: List[List[Dict[str, Any]]] = [] + for i, sq in enumerate(sub_queries): + try: + docs = self._first_stage(sq, base_table, retrieval_k, retrieval_mode, None) + except Exception as e: + print(f"โš ๏ธ Pooled first stage failed for sub-query {i + 1} ({e}); skipping it.") + docs = [] + per_sq.append(docs) + + pooled: List[Dict[str, Any]] = [] + by_key: Dict[Any, Dict[str, Any]] = {} + for rank in range(max((len(d) for d in per_sq), default=0)): + for i, docs in enumerate(per_sq): + if rank >= len(docs): + continue + doc = docs[rank] + key = (doc.get("document_id"), doc.get("chunk_index")) + if key in by_key and key != (None, None): + by_key[key].setdefault("_source_sqs", []).append(i) + continue + doc["_source_sqs"] = [i] + by_key[key] = doc + pooled.append(doc) - strategy = self.config.get("reranker", {}).get("strategy", "qwen") + print(f"๐Ÿงบ Pooled first stage: {sum(len(d) for d in per_sq)} docs from " + f"{len(sub_queries)} sub-queries โ†’ {len(pooled)} unique candidates.") + if event_callback: + event_callback("retrieval_done", {"count": len(pooled)}) + return pooled + + # Reranking (roadmap item 2.2: decomposition applies HERE, not first stage) + # ------------------------------------------------------------------ + @staticmethod + def _score_pairs(ai_reranker, strategy: str, query: str, texts: List[str]) -> Dict[int, float]: + """Score every candidate against one query. Returns {candidate index: score}.""" + # Some rerankers-lib backends are not thread-safe; serialise calls. + with _rerank_lock: if strategy == "rerankers-lib": - texts = [d['text'] for d in retrieved_docs] - # ColBERT's Rust backend isn't Sync; serialise calls. - with _rerank_lock: - ranked = ai_reranker.rank(query=query, docs=texts) - # ranked is RankedResults; convert to list of (score, idx) + ranked = ai_reranker.rank(query=query, docs=texts) try: pairs = [(r.score, r.document.doc_id) for r in ranked.results] - if any(p[1] is None for p in pairs): + if any(not isinstance(p[1], int) for p in pairs): pairs = [(r.score, i) for i, r in enumerate(ranked.results)] except Exception: pairs = ranked - # Keep only top_k results if requested - if top_k is not None and len(pairs) > top_k: - pairs = pairs[:top_k] - reranked_docs = [retrieved_docs[idx] | {"rerank_score": score} for score, idx in pairs] else: - try: - reranked_docs = ai_reranker.rerank(query, retrieved_docs, top_k=top_k) - except TypeError: - texts = [d['text'] for d in retrieved_docs] - pairs = ai_reranker.rank(query, texts, top_k=top_k) - reranked_docs = [retrieved_docs[idx] | {"rerank_score": score} for score, idx in pairs] - - rerank_time = time.time() - start_rerank_time - print(f"โœ… Reranking completed in {rerank_time:.2f}s. Refined to {len(reranked_docs)} docs.") + pairs = ai_reranker.rank(query, texts) + return {int(idx): float(score) for score, idx in pairs} + + def _rerank_stage(self, query: str, retrieved_docs: List[Dict[str, Any]], + sub_queries: Optional[List[str]] = None, + event_callback=None) -> List[Dict[str, Any]]: + """Reorder the first-stage candidates. No-op when reranking is off. + + When *sub_queries* is supplied (query decomposition is on **and** the + reranker is on), each candidate is scored against every sub-query and + the per-sub-query scores are aggregated with + ``query_decomposition.rerank_aggregate`` (``"mean"``, the default, or + ``"max"``). This is the whole of roadmap item 2.2: the first stage + always ran on the full original query, because decomposing *there* + dilutes the query semantically, while decomposition applied at reranking + is where the 2026 evidence puts the win. + """ + # Source tags from the pooled first stage (which sub-queries retrieved + # each candidate). Consumed here either way so they never leak into + # source_documents. + source_sqs = [doc.pop("_source_sqs", None) for doc in retrieved_docs] + + ai_reranker = self._get_ai_reranker() + if not ai_reranker or not retrieved_docs: + return retrieved_docs + + if event_callback: + event_callback("rerank_started", {"count": len(retrieved_docs)}) + print(f"\n--- Reranking top {len(retrieved_docs)} docs with AI model... ---") + start_rerank_time = time.time() + + rerank_cfg = self.config.get("reranker", {}) + top_k_cfg = rerank_cfg.get("top_k") + top_percent = rerank_cfg.get("top_percent") # value in range 0โ€“1 + + if top_percent is not None: + try: + pct = float(top_percent) + if not 0 < pct <= 1: + # Real validation, not an assert โ€” asserts vanish under + # `python -O` and a bad top_percent would slip through. + raise ValueError( + f"reranker.top_percent must be in (0, 1], got {top_percent!r}.") + top_k = max(1, int(len(retrieved_docs) * pct)) + except (TypeError, ValueError): + print("โš ๏ธ Invalid top_percent value; falling back to top_k") + top_k = top_k_cfg or len(retrieved_docs) + else: + top_k = top_k_cfg or len(retrieved_docs) + + strategy = rerank_cfg.get("strategy", "rerankers-lib") + # Score the chunk that matched retrieval, not the ยฑ1-merged block: the + # merge buries the matching chunk mid-string, beyond the scorer's + # truncation window, and the siblings dilute its relevance signal. + texts = [(d.get("metadata") or {}).get("core_text") or d["text"] + for d in retrieved_docs] + + sub_qs = [q for q in (sub_queries or []) if q and q.strip()] + # "mean" is the default because it measured better than "max" on both + # subsets of the item-2.2 A/B (eval/decisions/phase2-pipeline.md ยง3). + aggregate = (self.config.get("query_decomposition", {}) or {}).get( + "rerank_aggregate", "mean") + + pooled = any(source_sqs) and bool(sub_queries) + if pooled: + # Pooled mode: one pass, but each candidate is scored ONLY against + # the sub-queries that retrieved it โ€” the same pair count as the + # old per-sub-query rerank passes (minus dedupe), not + # candidates ร— queries. Indices in the tags refer to positions in + # the *original* sub_queries list, so no filtering/dedup here. + queries = list(sub_queries) + print(f"๐Ÿ”€ Pooled rerank: scoring each candidate against its source " + f"sub-quer(ies) of {len(queries)} (aggregate={aggregate}).") + per_query = [] + for qi, q in enumerate(queries): + idxs = [i for i, tags in enumerate(source_sqs) if tags and qi in tags] + if not idxs: + per_query.append({}) + continue + local = self._score_pairs(ai_reranker, strategy, q, [texts[i] for i in idxs]) + per_query.append({idxs[li]: s for li, s in local.items()}) + else: + queries = [query] + [q for q in sub_qs if q != query] if sub_qs else [query] + if len(queries) > 1: + print(f"๐Ÿ”€ Scoring candidates against {len(queries)} sub-queries " + f"(aggregate={aggregate}).") + per_query = [self._score_pairs(ai_reranker, strategy, q, texts) for q in queries] + + scores: Dict[int, float] = {} + for idx in range(len(texts)): + values = [m[idx] for m in per_query if idx in m] + if not values: + continue + scores[idx] = (sum(values) / len(values)) if aggregate == "mean" else max(values) + + ordered = sorted(scores.items(), key=lambda kv: kv[1], reverse=True) + + # Relevance-threshold SELECTION (not just reordering): keep only + # candidates whose best score against ANY query clears + # ``reranker.min_score`` โ€” union semantics, so a chunk that is + # relevant to one sub-query survives even if irrelevant to the rest. + # Only meaningful for the Qwen scorer, whose scores are calibrated + # P(relevant); raw cross-encoder logits have no fixed scale, so the + # threshold is skipped for them. ``min_keep`` floors the selection so + # an aggressive threshold can never empty the context. + min_score = rerank_cfg.get("min_score") + if min_score is not None and isinstance(ai_reranker, QwenRerankerScorer): + try: + min_score_value = float(min_score) + except (TypeError, ValueError): + raise ValueError( + f"reranker.min_score must be a number, got {min_score!r}.") + union_best = {} + for idx in scores: + union_best[idx] = max(m[idx] for m in per_query if idx in m) + kept = [(idx, s) for idx, s in ordered if union_best[idx] >= min_score_value] + if pooled: + # Every sub-question keeps at least its best candidate, or the + # single pooled synthesis silently loses that half of the + # question when nothing clears the threshold for it. + kept_idx = {idx for idx, _ in kept} + for qi in range(len(queries)): + if any(qi in (source_sqs[idx] or ()) for idx in kept_idx): + continue + cands = [(idx, s) for idx, s in ordered + if source_sqs[idx] and qi in source_sqs[idx]] + if cands: + kept.append(cands[0]) + kept_idx.add(cands[0][0]) + print(f"๐Ÿ›Ÿ Sub-query {qi + 1} kept its top candidate " + f"(score {cands[0][1]:.2f}) despite the threshold.") + kept.sort(key=lambda kv: kv[1], reverse=True) + min_keep = max(1, int(rerank_cfg.get("min_keep", 3))) + if len(kept) < min_keep: + # Union the floor with what survived โ€” including the + # per-sub-query rescues above โ€” instead of replacing it: + # `ordered[:min_keep]` alone would silently drop a rescued + # candidate that sits below the global top-min_keep. + floor_idx = {idx for idx, _ in kept} + kept = kept + [(idx, s) for idx, s in ordered[:min_keep] + if idx not in floor_idx] + kept.sort(key=lambda kv: kv[1], reverse=True) + if len(kept) != len(ordered): + print(f"๐ŸŽฏ Rerank threshold {min_score}: kept {len(kept)}/{len(ordered)} " + f"candidates (union-of-max across {len(queries)} query/queries).") + ordered = kept + + if top_k is not None and len(ordered) > top_k: + ordered = ordered[:top_k] + reranked_docs = [retrieved_docs[idx] | {"rerank_score": score} for idx, score in ordered] + + rerank_time = time.time() - start_rerank_time + print(f"โœ… Reranking completed in {rerank_time:.2f}s. Refined to {len(reranked_docs)} docs.") + if event_callback: + event_callback("rerank_done", {"count": len(reranked_docs)}) + return reranked_docs + + # ------------------------------------------------------------------ + # Cross-reference hop (roadmap item 4.2) + # ------------------------------------------------------------------ + + # Only the head of the candidate list gets to trigger a hop. A reference + # carried by candidate 17 is not evidence that the query is about the + # referenced document; it is evidence that candidate 17 is noise. + _CROSSREF_TRIGGER_DEPTH = 3 + + @staticmethod + def _chunk_crossrefs(doc: Dict[str, Any]) -> List[Dict[str, Any]]: + """The ``crossrefs`` list on a candidate, at either metadata nesting level. + + ``VectorIndexer`` stores ``json.dumps(chunk)`` in the ``metadata`` column, + so what the retriever hands back as ``doc["metadata"]`` is the whole chunk + dict with the real metadata nested one level down. Both shapes are + accepted so this keeps working if that ever gets straightened out. + """ + metadata = doc.get("metadata") + if not isinstance(metadata, dict): + return [] + refs = metadata.get("crossrefs") + if refs is None: + inner = metadata.get("metadata") + if isinstance(inner, dict): + refs = inner.get("crossrefs") + if not isinstance(refs, list): + return [] + return [r for r in refs if isinstance(r, dict)] + + def _crossref_hop(self, query: str, base_table: str, result: Dict[str, Any], + event_callback=None) -> Dict[str, Any]: + """One bounded hop from a top candidate's reference to the referenced doc. + + Runs only when ``retrieval.crossref_hop.enabled`` is true. No LLM, no + recursion: chunks pulled by a hop can never trigger another one, so the + worst case is ``max_hops`` extra filtered searches per query. + """ + cfg = self._crossref_hop_config() + if not cfg.get("enabled"): + return result + + documents = result.get("documents") or [] + max_hops = int(cfg.get("max_hops", 1) or 0) + chunks_per_hop = int(cfg.get("chunks_per_hop", 3) or 0) + if not documents or max_hops < 1 or chunks_per_hop < 1: + return result + + represented = {d.get("document_id") for d in documents} + targets: List[Dict[str, Any]] = [] + for rank, candidate in enumerate(documents[: self._CROSSREF_TRIGGER_DEPTH]): + for ref in self._chunk_crossrefs(candidate): + target = ref.get("target_doc") + if not target or target in represented: + continue + if any(t["target_doc"] == target for t in targets): + continue + targets.append({ + "target_doc": target, + "kind": ref.get("kind"), + "ref": ref.get("ref"), + "from_chunk_id": candidate.get("chunk_id"), + "from_document_id": candidate.get("document_id"), + "from_rank": rank, + }) + targets = targets[:max_hops] + if not targets: + return result + + hopped: List[Dict[str, Any]] = [] + for target in targets: + # Dense-only on purpose: the hop already knows *which* document it + # wants, so all that is left is picking its most on-topic chunks. + pulled = self._search_within_documents( + query, base_table, [target["target_doc"]], chunks_per_hop, + mode="vector_only") or [] + for doc in pulled: + doc["via_crossref"] = True + doc["crossref"] = {k: target[k] for k in + ("kind", "ref", "from_chunk_id", "from_document_id")} + if isinstance(doc.get("metadata"), dict): + doc["metadata"]["via_crossref"] = True + doc["metadata"]["crossref"] = doc["crossref"] + hopped.append(doc) + target["chunks_added"] = len(pulled) + + info = {"targets": targets, "chunks_added": len(hopped)} + if hopped: + # A new list: with reranking off, ``documents`` and ``first_stage`` + # are the same object, and the hop must not rewrite the first stage. + result["documents"] = list(documents) + hopped + result["crossref_hop"] = info + print(f"๐Ÿ”— Cross-reference hop: pulled {len(hopped)} chunk(s) from " + f"{len(targets)} referenced document(s) " + f"({', '.join(t['target_doc'] for t in targets)}).") + if event_callback: + event_callback("crossref_hop", info) + return result + + def _post_candidates(self, query: str, base_table: str, result: Dict[str, Any], + event_callback=None) -> Dict[str, Any]: + """Tail hook for every ``retrieve_candidates`` exit path. + + ``retrieve_candidates`` returns from five places (the retry has four + early outs). Anything that must run on a *final* candidate set goes here + once instead of being copy-pasted into each of them. + """ + return self._crossref_hop(query, base_table, result, event_callback) + + def retrieve_candidates(self, query: str, table_name: Optional[str] = None, + sub_queries: Optional[List[str]] = None, + event_callback=None, *, + filters: Any = None) -> Dict[str, Any]: + """First stage + rerank + the evidence-sufficiency retry around both. + + This is the whole candidate-selection path, factored out of ``run()`` so + that ``eval/run_eval.py`` measures exactly what ships rather than a + reimplementation of it. + + *filters* (roadmap item 4.4) is **keyword-only with a default**, so the + four positional arguments are exactly what they were and + ``EscalatingRetrievalPipeline``'s ``super()`` call is unaffected. It + accepts a raw filter object or an already-compiled one and raises + ``FilterError`` on anything invalid. When omitted, the thread-local + scope opened by ``run()`` is used โ€” which is how a filter reaches this + method through the escalation subclass at all. + + Returns:: + + {"first_stage": [...], # post-retry first-stage ordering + "documents": [...], # after reranking, or == first_stage + "query_used": str, + "retry": {...} | None, + # present only when a metadata filter was applied: + "filters": {"spec": {...}, "where": str}, + # present only when the cross-reference hop actually hopped: + "crossref_hop": {"targets": [...], "chunks_added": int}} + """ + retrieval_k = self.config.get("retrieval_k", 10) + retrieval_mode = self._retrieval_mode() + base_table = table_name or self.storage_config["text_table_name"] + + compiled = compile_filters(filters) if filters is not None else self.active_filter() + + if event_callback: + event_callback("retrieval_started", {"mode": retrieval_mode}) + + with self.filter_scope(compiled): + return self._retrieve_candidates_filtered( + query, base_table, sub_queries, event_callback, compiled, + retrieval_k, retrieval_mode) + + def _retrieve_candidates_filtered(self, query, base_table, sub_queries, event_callback, + compiled, retrieval_k, retrieval_mode) -> Dict[str, Any]: + """``retrieve_candidates``'s body, with the filter scope already open.""" + pooled = bool((self.config.get("query_decomposition") or {}).get("pooled_first_stage")) + if pooled and sub_queries and len(sub_queries) > 1: + first_stage = self._pooled_first_stage(sub_queries, base_table, retrieval_k, + retrieval_mode, event_callback) + else: + first_stage = self._first_stage(query, base_table, retrieval_k, retrieval_mode, + event_callback) + documents = self._rerank_stage(query, first_stage, sub_queries, event_callback) + + result = {"first_stage": first_stage, "documents": documents, + "query_used": query, "retry": None} + + if compiled is not None: + # Only present when a filter was actually applied, so an unfiltered + # result dict is byte-identical to what it was before item 4.4. + result["filters"] = {"spec": compiled.spec, "where": compiled.where} + print(f"๐Ÿ”Ž Metadata filter applied: {compiled.where} " + f"โ†’ {len(first_stage)} candidate(s).") if event_callback: - event_callback("rerank_done", {"count": len(reranked_docs)}) + event_callback("filters_applied", + {"spec": compiled.spec, "where": compiled.where, + "candidates": len(first_stage)}) + + retry_cfg = self._retry_config() + if not retry_cfg.get("enabled") or not first_stage: + return self._post_candidates(query, base_table, result, event_callback) + + # Prefer the reranker's calibrated probability when it produced one; + # otherwise fall back to the dense contrast score. RRF ranks are + # deliberately never used โ€” they carry no absolute information. + score = self._rerank_evidence_score(documents) + signal = "rerank" + threshold = retry_cfg.get("min_rerank_score", retry_cfg.get("min_top_score")) + if score is None: + score = self._dense_evidence_score(first_stage) + signal = "dense_contrast" + threshold = retry_cfg.get("min_top_score") + if score is None or threshold is None: + # fts_only, or a legacy unnormalized table: no meaningful signal. + return self._post_candidates(query, base_table, result, event_callback) + + if score >= float(threshold): + return self._post_candidates(query, base_table, result, event_callback) + + max_attempts = int(retry_cfg.get("max_attempts", 1) or 0) + if max_attempts < 1: + return self._post_candidates(query, base_table, result, event_callback) + + print(f"\n๐Ÿ” Evidence-sufficiency retry: {signal} score {score:.3f} " + f"< {float(threshold):.3f} โ€” reformulating once.") + reformulated = self._reformulate_query(query) + info = {"signal": signal, "threshold": float(threshold), + "score_before": round(score, 4), "reformulated": reformulated, + "attempted": reformulated is not None, "kept": "original", + "score_after": None} + + if reformulated: + retry_first = self._first_stage(reformulated, base_table, retrieval_k, + retrieval_mode, event_callback) + retry_docs = self._rerank_stage(reformulated, retry_first, sub_queries, + event_callback) + retry_score = (self._rerank_evidence_score(retry_docs) if signal == "rerank" + else self._dense_evidence_score(retry_first)) + info["score_after"] = None if retry_score is None else round(retry_score, 4) + # Keep whichever attempt scored better on the same signal. A retry + # that did not improve the evidence is discarded, not merged. + if retry_score is not None and retry_score > score: + result["first_stage"] = retry_first + result["documents"] = retry_docs + result["query_used"] = reformulated + info["kept"] = "retry" + + result["retry"] = info + if event_callback: + event_callback("retrieval_retry", info) + print(f"๐Ÿ” Retry kept the {info['kept']} result set " + f"(score_after={info['score_after']}).") + return self._post_candidates(query, base_table, result, event_callback) + + def run(self, query: str, table_name: str = None, window_size_override: Optional[int] = None, + event_callback=None, sub_queries: Optional[List[str]] = None, + *, filters: Any = None) -> Dict[str, Any]: + base_table = table_name or self.storage_config["text_table_name"] + + logger = logging.getLogger(__name__) + logger.debug("--- Running search for query '%s' (table=%s) ---", query, base_table) + + # A custom table_name is passed down the call chain (base_table goes to + # retrieve_candidates and _run_after_candidates) instead of being + # written back into self.storage_config: mutating shared config was + # sticky across queries and raced the agent's parallel sub-query fan-out. + + # Compiled here so an invalid filter raises before any retrieval work, + # and opened as a thread-local scope so it reaches `retrieve_candidates` + # without changing the positional signature the escalation subclass + # overrides. `filters=None` skips both and leaves this path untouched. + compiled = compile_filters(filters) + with self.filter_scope(compiled) if compiled is not None else contextlib.nullcontext(): + candidates = self.retrieve_candidates(query, base_table, sub_queries, event_callback) + return self._run_after_candidates(query, candidates, window_size_override, + event_callback, base_table) + + def _run_after_candidates(self, query: str, candidates: Dict[str, Any], + window_size_override: Optional[int], + event_callback, + base_table: Optional[str] = None) -> Dict[str, Any]: + """``run()``'s post-candidate half: expansion, pruning and synthesis.""" + reranked_docs = candidates["documents"] + + # Focus on the reranked chunks when the reranker ran โ€” applied to the + # SEED set, BEFORE expansion. Filtering the expanded set afterwards + # (the old behaviour) deleted every expanded neighbour, because only + # seeds carry a `rerank_score`, and silently nullified context + # expansion whenever the reranker was on. The cross-reference + # exemption is preserved: hops are appended *after* reranking by + # design (the reranker never saw them, and scoring them against the + # query would defeat the point: the referenced document is exactly + # the one whose text does not look like the query). + if any('rerank_score' in d for d in reranked_docs): + seed_docs = [d for d in reranked_docs + if 'rerank_score' in d or d.get('via_crossref')] else: - # If no AI reranker, proceed with the initially retrieved docs - reranked_docs = retrieved_docs + seed_docs = reranked_docs window_size = self.config.get("context_window_size", 1) if window_size_override is not None: window_size = window_size_override - if window_size > 0 and reranked_docs: + if window_size > 0 and seed_docs: if event_callback: - event_callback("context_expand_started", {"count": len(reranked_docs)}) - print(f"\n--- Expanding context for {len(reranked_docs)} top documents (window size: {window_size})... ---") - expanded_chunks = {} + event_callback("context_expand_started", {"count": len(seed_docs)}) + print(f"\n--- Expanding context for {len(seed_docs)} top documents (window size: {window_size})... ---") + # Capture the active metadata filter HERE, in the submitting + # thread: the executor's workers do not inherit thread-locals, so + # reading it inside the worker would silently drop the caller's + # filter (item 4.4) from the expansion queries. + active_filter = self.active_filter() + expanded_by_seed: Dict[int, List[Dict[str, Any]]] = {} with concurrent.futures.ThreadPoolExecutor() as executor: - future_to_chunk = {executor.submit(self._get_surrounding_chunks_lancedb, chunk, window_size): chunk for chunk in reranked_docs} - for future in concurrent.futures.as_completed(future_to_chunk): + future_to_seed = { + executor.submit(self._get_surrounding_chunks_lancedb, chunk, + window_size, table_name=base_table, + active_filter=active_filter): i + for i, chunk in enumerate(seed_docs) + } + for future in concurrent.futures.as_completed(future_to_seed): + seed_index = future_to_seed[future] try: - seed_chunk = future_to_chunk[future] - surrounding_chunks = future.result() - for surrounding_chunk in surrounding_chunks: - cid = surrounding_chunk['chunk_id'] - if cid not in expanded_chunks: - # If this is the *central* chunk we already reranked, carry over its score - if cid == seed_chunk.get('chunk_id') and 'rerank_score' in seed_chunk: - surrounding_chunk['rerank_score'] = seed_chunk['rerank_score'] - expanded_chunks[cid] = surrounding_chunk + expanded_by_seed[seed_index] = future.result() except Exception as e: print(f"Error expanding context for a chunk: {e}") + # Keep the seed itself rather than dropping the evidence. + expanded_by_seed[seed_index] = [seed_docs[seed_index]] - final_docs = list(expanded_chunks.values()) - # Sort by reranker score if present, otherwise by raw score/distance - if any('rerank_score' in d for d in final_docs): - final_docs.sort(key=lambda c: c.get('rerank_score', -1), reverse=True) - elif any('_distance' in d for d in final_docs): - # For vector search smaller distance is better - final_docs.sort(key=lambda c: c.get('_distance', 1e9)) - elif any('score' in d for d in final_docs): - final_docs.sort(key=lambda c: c.get('score', 0), reverse=True) - else: - # Fallback to document order - final_docs.sort(key=lambda c: (c.get('document_id', ''), c.get('chunk_index', 0))) + # Deterministic assembly: seeds in their incoming (rerank) score + # order, each seed's neighbours adjacent to it in chunk_index + # order. The first seed to claim a chunk_id wins the dedupe, so + # executor completion order never leaks into the result. + final_docs = [] + seen_chunk_ids = set() + for seed_index, seed_chunk in enumerate(seed_docs): + for surrounding_chunk in expanded_by_seed.get(seed_index, [seed_chunk]): + cid = surrounding_chunk.get('chunk_id') + if cid is not None: + if cid in seen_chunk_ids: + continue + seen_chunk_ids.add(cid) + is_seed = cid is not None and cid == seed_chunk.get('chunk_id') + # If this is the *central* chunk we already reranked, carry over its score + if is_seed and 'rerank_score' in seed_chunk: + surrounding_chunk['rerank_score'] = seed_chunk['rerank_score'] + # Same for the cross-reference marker: expansion re-reads + # the row from LanceDB, which knows nothing about how the + # chunk got here. + if is_seed and seed_chunk.get('via_crossref'): + surrounding_chunk['via_crossref'] = True + if seed_chunk.get('crossref'): + surrounding_chunk['crossref'] = seed_chunk['crossref'] + final_docs.append(surrounding_chunk) print(f"Expanded to {len(final_docs)} unique chunks for synthesis.") if event_callback: event_callback("context_expand_done", {"count": len(final_docs)}) else: - final_docs = reranked_docs - - # Optionally hide non-reranked chunks: if any chunk carries a - # `rerank_score`, we assume the caller wants to focus on those. - if any('rerank_score' in d for d in final_docs): - final_docs = [d for d in final_docs if 'rerank_score' in d] + final_docs = seed_docs # ------------------------------------------------------------------ # Sentence-level pruning (Provence) @@ -495,7 +1601,16 @@ def _clean_val(v): if key in doc: doc[key] = _clean_val(doc[key]) - context = "\n\n".join([doc['text'] for doc in final_docs]) + final_docs = self._budget_synthesis_context(final_docs) + # Label every snippet with its source document (arm I). The strict + # synthesis prompt forbids writing document names that are not in the + # snippets โ€” correct, but without labels the model *cannot* attribute + # a fact to its document even though the pipeline knows the source: + # corpora that cross-reference by tag (e.g. "[QUIC-TLS]") left answers + # attribution-blind ("the document" instead of the actual name). + context = "\n\n".join( + f"[Source document: {doc.get('document_id') or 'unknown'}]\n{doc['text']}" + for doc in final_docs) # ๐Ÿ‘€ DEBUG: Show the exact context passed to the LLM after pruning print("\n=== Context passed to LLM (post-pruning) ===") @@ -509,57 +1624,13 @@ def _clean_val(v): return {"answer": final_answer, "source_documents": final_docs} - # ------------------------------------------------------------------ - # Public utility - # ------------------------------------------------------------------ - def list_document_titles(self, max_items: int = 25) -> List[str]: - """Return up to *max_items* distinct document titles (or IDs). - - This is used only for prompt-routing, so we favour robustness over - perfect recall. If anything goes wrong we return an empty list so - the caller can degrade gracefully. - """ - try: - tbl_name = self.storage_config.get("text_table_name") - if not tbl_name: - return [] - - tbl = self._get_db_manager().get_table(tbl_name) - - field_name = "document_title" if "document_title" in tbl.schema.names else "document_id" - - # Use a cheap SQL filter to grab distinct values; fall back to a - # simple scan if the driver lacks DISTINCT support. - try: - sql = f"SELECT DISTINCT {field_name} FROM tbl LIMIT {max_items}" - rows = tbl.search().where("true").sql(sql).to_list() # type: ignore - titles = [r[field_name] for r in rows if r.get(field_name)] - except Exception: - # Fallback: scan first N rows - rows = tbl.search().select(field_name).limit(max_items * 4).to_list() - seen = set() - titles = [] - for r in rows: - val = r.get(field_name) - if val and val not in seen: - titles.append(val) - seen.add(val) - if len(titles) >= max_items: - break - - # Ensure we don't exceed max_items - return titles[:max_items] - except Exception: - # Any issues (missing table, bad schema, etc.) โ€“> just return [] - return [] - # -------------------- Public helper properties -------------------- @property def retriever(self): - """Lazily exposes the main (dense) retriever so external components - like the ReAct agent tools can call `.retrieve()` directly without - reaching into private helpers. If the retriever has not yet been - instantiated, it is created on first access via `_get_dense_retriever`.""" + """Lazily exposes the MultiVectorRetriever so external components can + call `.retrieve()` directly without reaching into private helpers. If + the retriever has not yet been instantiated, it is created on first + access via `_get_dense_retriever`.""" return self._get_dense_retriever() def update_embedding_model(self, model_name: str): @@ -570,4 +1641,5 @@ def update_embedding_model(self, model_name: str): self.config["embedding_model_name"] = model_name # Reset caches so new instances are built on demand self.text_embedder = None - self.dense_retriever = None \ No newline at end of file + self.dense_retriever = None + self._dense_retriever_error = None \ No newline at end of file diff --git a/rag_system/requirements.txt b/rag_system/requirements.txt index 1387f755..be434807 100644 --- a/rag_system/requirements.txt +++ b/rag_system/requirements.txt @@ -1,15 +1,10 @@ -colpali-engine PyMuPDF Pillow transformers==4.51.0 torch==2.4.1 torchvision==0.19.1 -lancedb -rank_bm25 -fuzzywuzzy -python-Levenshtein torchaudio -transformers +lancedb sentencepiece accelerate docling diff --git a/rag_system/rerankers/reranker.py b/rag_system/rerankers/reranker.py index 54332d36..3fbd2fe7 100644 --- a/rag_system/rerankers/reranker.py +++ b/rag_system/rerankers/reranker.py @@ -1,12 +1,12 @@ -from transformers import AutoModelForSequenceClassification, AutoTokenizer +from transformers import AutoModelForCausalLM, AutoModelForSequenceClassification, AutoTokenizer import torch -from typing import List, Dict, Any +from typing import List, Optional, Tuple -class QwenReranker: +class CrossEncoderReranker: """ - A reranker that uses a local Hugging Face transformer model. + A cross-encoder reranker backed by a local Hugging Face sequence-classification model. """ - def __init__(self, model_name: str = "BAAI/bge-reranker-base"): + def __init__(self, model_name: str = "BAAI/bge-reranker-v2-m3"): # Auto-select the best available device: CUDA > MPS > CPU if torch.cuda.is_available(): self.device = "cuda" @@ -14,40 +14,43 @@ def __init__(self, model_name: str = "BAAI/bge-reranker-base"): self.device = "mps" else: self.device = "cpu" - print(f"Initializing BGE Reranker with model '{model_name}' on device '{self.device}'.") + print(f"Initializing cross-encoder reranker with model '{model_name}' on device '{self.device}'.") self.tokenizer = AutoTokenizer.from_pretrained(model_name) self.model = AutoModelForSequenceClassification.from_pretrained( model_name, torch_dtype=torch.float16 if self.device != "cpu" else None, ).to(self.device).eval() - print("BGE Reranker loaded successfully.") + print("Cross-encoder reranker loaded successfully.") - def _format_instruction(self, query: str, doc: str): - instruction = 'Given a web search query, retrieve relevant passages that answer the query' - return f"<Instruct>: {instruction}\n<Query>: {query}\n<Document>: {doc}" + def rank(self, query: str, texts: List[str], *, early_exit: bool = True, margin: float = 0.4, min_scored: int = 8, batch_size: int = 8) -> List[Tuple[float, int]]: + """Score every text against *query*. - def rerank(self, query: str, documents: List[Dict[str, Any]], top_k: int = 5, *, early_exit: bool = True, margin: float = 0.4, min_scored: int = 8, batch_size: int = 8) -> List[Dict[str, Any]]: - """ - Reranks a list of documents based on their relevance to a query. + Returns ``[(score, original_index), โ€ฆ]`` sorted by score, descending โ€” + the same interface ``QwenRerankerScorer.rank`` and the `rerankers` lib + branch expose, so ``RetrievalPipeline._score_pairs`` can call any of + them. Scores are raw logits. - If *early_exit* is True the cross-encoder scores documents in mini-batches and - stops once the best-so-far score beats the worst-so-far by *margin* after at - least *min_scored* docs have been processed. This accelerates "easy" queries - where strong positives dominate. + If *early_exit* is True the cross-encoder scores documents in + mini-batches and stops once the best-so-far score beats the + worst-so-far by *margin* after at least *min_scored* docs have been + processed. The margin is compared on **sigmoid-normalized** scores: + raw logits routinely spread wider than the default margin after a + single batch, which made the check fire immediately and cut scoring + off after one batch. This accelerates "easy" queries where strong + positives dominate. """ - if not documents: + if not texts: return [] - # Sort by the upstream (hybrid) score so that the strongest candidates are evaluated first. - docs_sorted = sorted(documents, key=lambda d: d.get('score', 0.0), reverse=True) - - scored_pairs: List[tuple[float, Dict[str, Any]]] = [] + # Candidates arrive in upstream (hybrid) rank order, so the strongest + # are evaluated first and the early exit actually saves work. + scored_pairs: List[Tuple[float, int]] = [] + normed_scores: List[float] = [] with torch.no_grad(): - for start in range(0, len(docs_sorted), batch_size): - batch_docs = docs_sorted[start : start + batch_size] - batch_pairs = [[query, d['text']] for d in batch_docs] + for start in range(0, len(texts), batch_size): + batch_pairs = [[query, text] for text in texts[start : start + batch_size]] inputs = self.tokenizer( batch_pairs, @@ -57,49 +60,188 @@ def rerank(self, query: str, documents: List[Dict[str, Any]], top_k: int = 5, *, max_length=512, ).to(self.device) - logits = self.model(**inputs).logits.view(-1) - batch_scores = logits.float().cpu().tolist() - - scored_pairs.extend(zip(batch_scores, batch_docs)) + logits = self.model(**inputs).logits.view(-1).float() + scored_pairs.extend(zip(logits.cpu().tolist(), + range(start, start + len(batch_pairs)))) + normed_scores.extend(torch.sigmoid(logits).cpu().tolist()) - # --- Early-exit check --- + # --- Early-exit check (on the normalized scores) --- if early_exit and len(scored_pairs) >= min_scored: - # Current best and worst among *already* scored docs - best_score = max(scored_pairs, key=lambda x: x[0])[0] - worst_score = min(scored_pairs, key=lambda x: x[0])[0] - if best_score - worst_score >= margin: + if max(normed_scores) - min(normed_scores) >= margin: break - # Sort final set and attach scores - sorted_by_score = sorted(scored_pairs, key=lambda x: x[0], reverse=True) - reranked_docs: List[Dict[str, Any]] = [] - for score, doc in sorted_by_score[:top_k]: - doc_with_score = doc.copy() - doc_with_score['rerank_score'] = score - reranked_docs.append(doc_with_score) + scored_pairs.sort(key=lambda pair: pair[0], reverse=True) + return scored_pairs + +def is_qwen3_reranker(model_name: str) -> bool: + """True for the Qwen3-Reranker family (causal-LM yes/no scorers).""" + return "qwen3-reranker" in (model_name or "").lower() + + +class QwenRerankerScorer: + """ + Reranker for the ``Qwen/Qwen3-Reranker-*`` family. + + These are **causal LMs**, not ``AutoModelForSequenceClassification`` models. + Loading them through the `rerankers` library's cross-encoder path builds a + ``Qwen3ForSequenceClassification`` with a **randomly initialised** ``score`` + head, which produces meaningless (untrained) scores. This class implements + the scoring scheme published on the model card instead: the query/document + pair is wrapped in the model's chat template, the model is asked whether the + document satisfies the query, and the score is the probability of the "yes" + token against the "no" token at the final position. + + Interface mirrors what ``RetrievalPipeline`` and ``eval/run_eval.py`` expect + from the `rerankers` lib branch: ``rank(query=..., docs=[...])`` returns a + list of ``(score, original_index)`` tuples sorted by score, descending. + """ + + DEFAULT_INSTRUCTION = ( + "Given a web search query, retrieve relevant passages that answer the query" + ) + PREFIX = ( + "<|im_start|>system\nJudge whether the Document meets the requirements " + 'based on the Query and the Instruct provided. Note that the answer can ' + 'only be "yes" or "no".<|im_end|>\n<|im_start|>user\n' + ) + SUFFIX = "<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n" + + def __init__( + self, + model_name: str = "Qwen/Qwen3-Reranker-0.6B", + *, + instruction: Optional[str] = None, + max_length: int = 2048, + batch_size: int = 8, + device: Optional[str] = None, + ): + if device: + self.device = device + elif torch.cuda.is_available(): + self.device = "cuda" + elif getattr(torch.backends, "mps", None) and torch.backends.mps.is_available(): + self.device = "mps" + else: + self.device = "cpu" + + self.model_name = model_name + self.instruction = instruction or self.DEFAULT_INSTRUCTION + self.max_length = max_length + self.batch_size = batch_size + + print(f"Initializing Qwen3 reranker '{model_name}' on device '{self.device}'.") + self.tokenizer = AutoTokenizer.from_pretrained(model_name, padding_side="left") + self.model = AutoModelForCausalLM.from_pretrained( + model_name, + torch_dtype=torch.float16 if self.device != "cpu" else torch.float32, + ).to(self.device).eval() + + self.token_true_id = self.tokenizer.convert_tokens_to_ids("yes") + self.token_false_id = self.tokenizer.convert_tokens_to_ids("no") + self.prefix_tokens = self.tokenizer.encode(self.PREFIX, add_special_tokens=False) + self.suffix_tokens = self.tokenizer.encode(self.SUFFIX, add_special_tokens=False) + print("Qwen3 reranker loaded successfully.") + + # -- internals --------------------------------------------------------- + + def _format_pair(self, query: str, doc: str) -> str: + return f"<Instruct>: {self.instruction}\n<Query>: {query}\n<Document>: {doc}" + + def _score_batch(self, pairs: List[str]) -> List[float]: + budget = self.max_length - len(self.prefix_tokens) - len(self.suffix_tokens) + enc = self.tokenizer( + pairs, + padding=False, + truncation="longest_first", + return_attention_mask=False, + max_length=max(budget, 16), + ) + enc["input_ids"] = [ + self.prefix_tokens + ids + self.suffix_tokens for ids in enc["input_ids"] + ] + enc = self.tokenizer.pad(enc, padding=True, return_tensors="pt") + enc = {k: v.to(self.device) for k, v in enc.items()} + + with torch.no_grad(): + logits = self.model(**enc).logits[:, -1, :].float() + stacked = torch.stack( + [logits[:, self.token_false_id], logits[:, self.token_true_id]], dim=1 + ) + probs = torch.nn.functional.log_softmax(stacked, dim=1)[:, 1].exp() + return probs.cpu().tolist() + + # -- public API -------------------------------------------------------- + + def score(self, query: str, docs: List[str]) -> List[float]: + """Relevance probability in [0, 1], one per document, in input order.""" + scores: List[float] = [] + for start in range(0, len(docs), self.batch_size): + batch = docs[start : start + self.batch_size] + scores.extend(self._score_batch([self._format_pair(query, d) for d in batch])) + return scores + + def rank(self, query: str, docs: List[str], top_k: Optional[int] = None + ) -> List[Tuple[float, int]]: + """``[(score, original_index), โ€ฆ]`` sorted by score, descending.""" + if not docs: + return [] + scores = self.score(query, docs) + pairs = sorted(enumerate(scores), key=lambda x: x[1], reverse=True) + out = [(score, idx) for idx, score in pairs] + return out[:top_k] if top_k else out - return reranked_docs if __name__ == '__main__': # This test requires an internet connection to download the models. try: - reranker = QwenReranker(model_name="BAAI/bge-reranker-base") - + reranker = CrossEncoderReranker(model_name="BAAI/bge-reranker-v2-m3") + query = "What is the capital of France?" documents = [ - {'text': "Paris is the capital of France.", 'metadata': {'doc_id': 'a'}}, - {'text': "The Eiffel Tower is in Paris.", 'metadata': {'doc_id': 'b'}}, - {'text': "France is a country in Europe.", 'metadata': {'doc_id': 'c'}}, + "Paris is the capital of France.", + "The Eiffel Tower is in Paris.", + "France is a country in Europe.", ] - - reranked_documents = reranker.rerank(query, documents) - + + ranked = reranker.rank(query, documents) + print("\n--- Verification ---") print(f"Query: {query}") print("Reranked documents:") - for doc in reranked_documents: - print(f" - Score: {doc['rerank_score']:.4f}, Text: {doc['text']}") + for score, idx in ranked: + print(f" - Score: {score:.4f}, Text: {documents[idx]}") except Exception as e: - print(f"\nAn error occurred during the QwenReranker test: {e}") + print(f"\nAn error occurred during the CrossEncoderReranker test: {e}") print("Please ensure you have an internet connection for model downloads.") + + +class MaxSimRerankerScorer: + """Late-interaction (ColBERT/MaxSim) rescorer via a local scoring sidecar. + + Experimental (component ablation follow-up, 2026-08-18): a 149M + multi-vector model (lightonai/LateOn) as a drop-in replacement for the 4B + cross-encoder. Runs OUT OF PROCESS: SentenceTransformers v6 requires + transformers 5.x / torch >= 2.5, which conflicts with this repo's pinned + MPS stack, so the model lives in its own venv behind a localhost endpoint + (scratch maxsim/server.py during the experiment). + + Interface matches the non-library branch of ``_score_pairs``: + ``rank(query, docs)`` returns ``[(score, original_index)]`` sorted by + score, descending. Scores are raw MaxSim sums (rank-meaningful, NOT + calibrated probabilities), so the pipeline's ``min_score`` threshold โ€” + gated on ``isinstance(..., QwenRerankerScorer)`` โ€” deliberately does not + apply; selection is plain top_k. + """ + + def __init__(self, endpoint: str | None = None): + import os + self.endpoint = endpoint or os.getenv("MAXSIM_ENDPOINT", "http://127.0.0.1:8765") + + def rank(self, query: str, docs): + import requests as _requests + resp = _requests.post(self.endpoint, json={"query": query, "docs": list(docs)}, timeout=120) + resp.raise_for_status() + scores = resp.json()["scores"] + pairs = sorted(((float(s), i) for i, s in enumerate(scores)), reverse=True) + return pairs diff --git a/rag_system/retrieval/document_fetch.py b/rag_system/retrieval/document_fetch.py new file mode 100644 index 00000000..84a4609a --- /dev/null +++ b/rag_system/retrieval/document_fetch.py @@ -0,0 +1,251 @@ +"""Reassemble a whole document from its indexed chunks (roadmap item 4.1). + +This is the "deep read" half of full-document escalation: given any chunk's +``document_id``, pull every chunk of that document out of the LanceDB text table +and glue them back together **in ``chunk_index`` order**, capped at a token +budget. Chunk order is the point โ€” DOS-RAG's finding is that handing a model the +document in its original order beats handing it the same text ranked by +similarity โ€” so a document whose chunks cannot be ordered is not escalated at +all rather than escalated scrambled. + +Nothing here calls an LLM or an embedder. It is a metadata filter and a sort. + +Token counting is deliberately crude: ``len(text) // 4``. The exact number does +not need to be right, it needs to be *cheap* and never to under-count badly +enough to blow a context window; a 4-chars-per-token estimate runs slightly +conservative on English prose and needs no tokenizer load. It is reported as +``approx_tokens`` everywhere so no caller mistakes it for a real count. +""" + +from __future__ import annotations + +import json +import os +import uuid +from dataclasses import dataclass, field +from typing import Any, Dict, List, Optional + +# Chunks are a few hundred tokens each; this is a hard stop on how many rows we +# will pull for one document, not an expected value. When the cap engages the +# reassembly is flagged ``truncated`` โ€” silently pretending the document was +# whole would be worse than saying part of it is missing. +_MAX_CHUNKS_SCANNED = 2000 + +_CHARS_PER_TOKEN = 4 + + +def approximate_token_count(text: str) -> int: + """A tokenizer-free token estimate: ``len(text) // 4``. + + See the module docstring for why this is not a real tokenizer count. + """ + if not text: + return 0 + return len(text) // _CHARS_PER_TOKEN + + +@dataclass +class FetchedDocument: + """One reassembled document, already truncated to the caller's budget.""" + + document_id: str + document_name: str + text: str + chunks_used: int + chunks_total: int + approx_tokens: int + truncated: bool + chunk_indices: List[int] = field(default_factory=list) + + def as_event_payload(self) -> Dict[str, Any]: + """The subset worth putting on the SSE wire (never the document text).""" + return { + "document_id": self.document_id, + "document_name": self.document_name, + "chunks_used": self.chunks_used, + "chunks_total": self.chunks_total, + "approx_tokens": self.approx_tokens, + "truncated": self.truncated, + } + + +def _parse_metadata(raw: Any) -> Dict[str, Any]: + """The ``metadata`` column stores the whole chunk dict as a JSON string.""" + if isinstance(raw, dict): + return raw + if isinstance(raw, str) and raw: + try: + parsed = json.loads(raw) + return parsed if isinstance(parsed, dict) else {} + except json.JSONDecodeError: + return {} + return {} + + +def _chunk_text(row: Dict[str, Any]) -> str: + """Prefer the clean ``original_text`` over the enriched/contextualised text. + + Indexing stores the chunk dict as JSON in the ``metadata`` column, so the + clean text sits at ``metadata["metadata"]["original_text"]``. Older or + hand-built rows put it one level up. The top-level ``text`` column is the + last resort: with contextual enrichment on it carries a prepended + "Context: โ€ฆ" summary, which is noise when the whole document is present. + """ + meta = _parse_metadata(row.get("metadata")) + inner = meta.get("metadata") if isinstance(meta.get("metadata"), dict) else {} + for candidate in (inner.get("original_text"), meta.get("original_text"), row.get("text")): + if isinstance(candidate, str) and candidate.strip(): + return candidate + return "" + + +def _document_name(document_id: str, rows: List[Dict[str, Any]]) -> str: + """A human-readable name: the indexed ``source`` path's basename.""" + for row in rows: + meta = _parse_metadata(row.get("metadata")) + inner = meta.get("metadata") if isinstance(meta.get("metadata"), dict) else {} + source = inner.get("source") or meta.get("source") + if isinstance(source, str) and source.strip(): + return os.path.basename(source.strip()) + # document_id is "<uuid>_<filename>" for files uploaded through the UI, + # but the plain basename for CLI-indexed files โ€” and a basename can itself + # contain underscores ("07_nda.pdf"). Strip the prefix only when it is + # actually a UUID; otherwise the whole id is the best name we have. + prefix, sep, tail = document_id.partition("_") + if sep and tail: + try: + uuid.UUID(prefix) + return tail + except ValueError: + pass + return document_id + + +def fetch_document( + db_manager, + table_name: str, + document_id: str, + token_budget: int = 6000, +) -> Optional[FetchedDocument]: + """Reassemble ``document_id`` from ``table_name``, in chunk order. + + Returns ``None`` โ€” never raises โ€” when the table cannot be opened, the + document has no rows, or the rows carry no usable ``chunk_index``. The + caller treats that as "no escalation", which is the safe outcome. + """ + if not document_id or not table_name: + return None + + # The filter is interpolated into a SQL string, so refuse an id that could + # terminate the literal instead of trying to escape it. + if "'" in document_id or "\\" in document_id: + print(f"โš ๏ธ Refusing to escalate document id with quoting characters: {document_id!r}") + return None + + try: + table = db_manager.get_table(table_name) + except Exception as e: + print(f"โš ๏ธ Full-document escalation could not open table '{table_name}': {e}") + return None + + try: + rows = ( + table.search() + .where(f"document_id = '{document_id}'") + .limit(_MAX_CHUNKS_SCANNED) + .to_list() + ) + except Exception as e: + print(f"โš ๏ธ Full-document escalation query failed for '{document_id}': {e}") + return None + + if not rows: + return None + + # Hitting the scan cap means the document may hold more chunks than we + # pulled; the reassembly is then truncated even when the token budget + # below never engages. + hit_scan_cap = len(rows) >= _MAX_CHUNKS_SCANNED + + # Order is the whole point of this module: no order, no escalation. + ordered: Dict[int, Dict[str, Any]] = {} + for row in rows: + index = row.get("chunk_index") + if index is None or index == -1: + continue + try: + index = int(index) + except (TypeError, ValueError): + continue + # A duplicate chunk_index means two rows claim the same slot (e.g. a + # late-chunk table merged in). Keep the first and move on. + ordered.setdefault(index, row) + + if not ordered: + print(f"โš ๏ธ Document '{document_id}' has no ordered chunks; skipping escalation.") + return None + + chunks_total = len(ordered) + budget_chars = max(0, int(token_budget)) * _CHARS_PER_TOKEN + + pieces: List[str] = [] + used_indices: List[int] = [] + length = 0 + truncated = False + + for index in sorted(ordered): + text = _chunk_text(ordered[index]).strip() + if not text: + continue + separator = 2 if pieces else 0 # the "\n\n" join + remaining = budget_chars - length - separator + if remaining <= 0: + truncated = True + break + if len(text) > remaining: + # Cut mid-chunk rather than dropping the chunk: the budget is a + # context-window guard, and a partial final chunk still reads in order. + pieces.append(text[:remaining]) + used_indices.append(index) + length += remaining + separator + truncated = True + break + pieces.append(text) + used_indices.append(index) + length += len(text) + separator + + if not pieces: + return None + + body = "\n\n".join(pieces) + return FetchedDocument( + document_id=document_id, + document_name=_document_name(document_id, rows), + text=body, + chunks_used=len(used_indices), + chunks_total=chunks_total, + approx_tokens=approximate_token_count(body), + truncated=truncated or hit_scan_cap, + chunk_indices=used_indices, + ) + + +def format_escalation_block(document: FetchedDocument) -> str: + """Wrap a reassembled document in the delimiter synthesis sees. + + Kept clearly separate from the retrieved snippets so the model can tell the + two apart, and labelled "escalated" so the answer's sourcing story stays + honest: the chunk citations are still the citations, this block is extra + reading material. + """ + header = f"FULL DOCUMENT (escalated): {document.document_name}" + if document.truncated: + header += ( + f" [truncated to ~{document.approx_tokens} tokens; " + f"{document.chunks_used} of {document.chunks_total} chunks]" + ) + return ( + "โ€“โ€“โ€“โ€“โ€“ " + header + " โ€“โ€“โ€“โ€“โ€“\n" + f"{document.text}\n" + "โ€“โ€“โ€“โ€“โ€“ END FULL DOCUMENT โ€“โ€“โ€“โ€“โ€“" + ) diff --git a/rag_system/retrieval/filters.py b/rag_system/retrieval/filters.py new file mode 100644 index 00000000..948b0bc7 --- /dev/null +++ b/rag_system/retrieval/filters.py @@ -0,0 +1,296 @@ +"""A deterministic metadata filter DSL that compiles to LanceDB SQL (roadmap 4.4). + +The surface the roadmap asks for is ``field=value`` / ``field in (a, b)``. Here it +is JSON rather than a filter *string*, for one reason: a string has to be parsed, +and a parser is the thing an injection attack aims at. A JSON object is already +parsed by the time it reaches us, so the only work left is **validation**, and +validation is the whole security story of this module. + + {"document_id": "07_nda.pdf"} + {"document_id": {"in": ["07_nda.pdf", "03_ip_certification.pdf"]}} + {"document_name": {"contains": "nda"}, "chunk_index": {"gte": 0, "lte": 4}} + +Top-level keys are ANDed. There is no OR, no NOT and no nesting โ€” not because +they are hard, but because nothing has asked for them and every operator added +here is another string that ends up inside a SQL predicate. + +No LLM is involved. The same filter object always compiles to the same +where-clause, byte for byte (fields are emitted in a fixed canonical order, not +in dict order), so a filtered retrieval is as reproducible as an unfiltered one. + +Rules this module holds to +-------------------------- +* **Refuse, don't escape.** Following ``rag_system/retrieval/document_fetch.py``: + a value carrying a quote, a backslash or a control character is rejected with + a ``FilterError``, never repaired. The one exception is the ``LIKE`` wildcards + ``%`` and ``_``, which are escaped (with an explicit ``ESCAPE '\\'`` clause) so + that ``contains`` means *substring*, literally, for filenames like + ``01_acquisition_agreement.pdf``. +* **Fail loud.** Unknown field, unknown operator, wrong value type, empty filter + โ€” all raise ``FilterError``. The API turns that into a 400. A filter that + cannot be honoured must never degrade into an unfiltered search: that would + hand the caller results they explicitly excluded. +* **Only real columns.** The LanceDB text table has exactly six columns + (``vector``, ``text``, ``chunk_id``, ``document_id``, ``chunk_index``, + ``metadata``) and ``metadata`` is a JSON *string*. So page numbers, dates and + every other per-chunk field are **not** filterable here; see the decision file + for the schema change that would fix it. Filtering on a JSON substring would + look like it worked and quietly be wrong. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Any, Dict, List, Optional, Tuple + +# A LanceDB int32 column; a value outside this cannot match anything and is far +# more likely to be a client bug than an intention. +_INT32_MAX = 2**31 - 1 +_INT32_MIN = -(2**31) + +# Long enough for any real document id (uuid + filename), short enough that a +# filter cannot be used to push a megabyte of text into a query plan. +_MAX_STRING_LENGTH = 256 + +# An IN-list this long is a client that means "no filter". +_MAX_IN_ITEMS = 256 + +# Characters that could terminate or extend the SQL literal we are building. +# Rejected, not escaped โ€” see the module docstring. +_FORBIDDEN_CHARS = ("'", '"', "\\", ";", "`", "\n", "\r", "\t", "\x00") + + +class FilterError(ValueError): + """A filter that cannot be compiled. Callers must surface this, not swallow it.""" + + +@dataclass(frozen=True) +class CompiledFilter: + """A validated filter plus the SQL it compiles to. + + ``where`` is the only thing that reaches LanceDB. ``spec`` is the caller's + original object, kept for logging, SSE payloads and the semantic-cache key โ€” + two queries with different filters are different queries. + """ + + where: str + spec: Dict[str, Any] + + def __str__(self) -> str: # pragma: no cover - convenience only + return self.where + + @property + def signature(self) -> str: + """A stable identity for this filter, for cache keys.""" + return self.where + + +# -------------------------------------------------------------------------- +# field table +# -------------------------------------------------------------------------- +# +# (column, kind, allowed operators). Order matters: it is the canonical emission +# order, so the compiled where-clause does not depend on JSON key order. + +_STRING_OPS = ("eq", "in", "contains") +_INT_OPS = ("eq", "in", "gt", "gte", "lt", "lte") + +_FIELDS: Tuple[Tuple[str, str, str, Tuple[str, ...]], ...] = ( + # name column kind operators + ("document_id", "document_id", "string", _STRING_OPS), + # `document_name` is not a column. Document ids are the file's basename for + # anything indexed by the CLI and "<uuid>_<basename>" for anything uploaded + # through the UI, so matching a *name* means substring-matching the id. + # Only `contains` is offered, because `eq` would silently miss every + # UI-uploaded document and that is a trap, not a feature. + ("document_name", "document_id", "string", ("contains",)), + ("chunk_id", "chunk_id", "string", ("eq", "in")), + ("chunk_index", "chunk_index", "int", _INT_OPS), +) + +_FIELD_BY_NAME = {name: (column, kind, ops) for name, column, kind, ops in _FIELDS} + +_COMPARISON_SQL = {"gt": ">", "gte": ">=", "lt": "<", "lte": "<="} + + +def _describe_support() -> str: + return "; ".join(f"{name} ({', '.join(ops)})" for name, _c, _k, ops in _FIELDS) + + +# -------------------------------------------------------------------------- +# value validation +# -------------------------------------------------------------------------- + +def _check_string(field: str, operator: str, value: Any) -> str: + if not isinstance(value, str): + raise FilterError( + f"filters.{field}.{operator} must be a string, got {type(value).__name__}." + ) + if not value.strip(): + raise FilterError(f"filters.{field}.{operator} must not be empty.") + if len(value) > _MAX_STRING_LENGTH: + raise FilterError( + f"filters.{field}.{operator} is longer than {_MAX_STRING_LENGTH} characters." + ) + for char in _FORBIDDEN_CHARS: + if char in value: + raise FilterError( + f"filters.{field}.{operator} contains a forbidden character " + f"({char!r}); quoting characters are refused, not escaped." + ) + # Anything else non-printable would be invisible in a log line. + if any(ord(c) < 0x20 or ord(c) == 0x7F for c in value): + raise FilterError( + f"filters.{field}.{operator} contains a control character." + ) + return value + + +def _check_int(field: str, operator: str, value: Any) -> int: + # bool is an int subclass in Python; True as a chunk_index is a client bug. + if isinstance(value, bool) or not isinstance(value, int): + raise FilterError( + f"filters.{field}.{operator} must be an integer, got {type(value).__name__}." + ) + if not (_INT32_MIN <= value <= _INT32_MAX): + raise FilterError(f"filters.{field}.{operator} is out of the int32 range.") + return value + + +def _quote(value: str) -> str: + """Wrap an already-validated string in SQL single quotes. + + Safe only because ``_check_string`` has refused every quoting character; this + function deliberately does no escaping of its own, so that the validation + above stays the single place where trust is granted. + """ + return "'" + value + "'" + + +def _like_pattern(value: str) -> str: + """``%value%`` with the LIKE wildcards inside *value* escaped.""" + escaped = value.replace("%", "\\%").replace("_", "\\_") + return "'%" + escaped + "%'" + + +# -------------------------------------------------------------------------- +# compilation +# -------------------------------------------------------------------------- + +def _compile_field(field: str, column: str, kind: str, + allowed: Tuple[str, ...], constraint: Any) -> List[str]: + """The SQL predicates for one field. ANDed together by the caller.""" + # A bare scalar is shorthand for equality โ€” the roadmap's `field=value`. + if not isinstance(constraint, dict): + constraint = {"eq": constraint} + if not constraint: + raise FilterError(f"filters.{field} has no operators.") + + predicates: List[str] = [] + for operator in sorted(constraint): # canonical order, not dict order + if operator not in allowed: + raise FilterError( + f"filters.{field} does not support the '{operator}' operator. " + f"Supported: {', '.join(allowed)}." + ) + value = constraint[operator] + + if operator == "in": + if not isinstance(value, list): + raise FilterError(f"filters.{field}.in must be a list.") + if not value: + raise FilterError( + f"filters.{field}.in is an empty list; an IN-list that " + "matches nothing is refused rather than silently dropped." + ) + if len(value) > _MAX_IN_ITEMS: + raise FilterError( + f"filters.{field}.in has {len(value)} items (max {_MAX_IN_ITEMS})." + ) + if kind == "string": + items = [_quote(_check_string(field, "in", v)) for v in value] + else: + items = [str(_check_int(field, "in", v)) for v in value] + predicates.append(f"{column} IN ({', '.join(items)})") + + elif operator == "eq": + if kind == "string": + predicates.append(f"{column} = {_quote(_check_string(field, 'eq', value))}") + else: + predicates.append(f"{column} = {_check_int(field, 'eq', value)}") + + elif operator == "contains": + pattern = _like_pattern(_check_string(field, "contains", value)) + # The ESCAPE clause is what makes '_' in "01_acquisition.pdf" a + # literal underscore instead of "any single character". + predicates.append(f"{column} LIKE {pattern} ESCAPE '\\'") + + else: # gt / gte / lt / lte + number = _check_int(field, operator, value) + predicates.append(f"{column} {_COMPARISON_SQL[operator]} {number}") + + return predicates + + +def compile_filters(filters: Any) -> Optional[CompiledFilter]: + """Validate and compile a filter object into a LanceDB where-clause. + + Returns ``None`` for ``None`` (the no-filter path, which must stay + byte-identical to having no filter support at all) and a ``CompiledFilter`` + otherwise. Raises ``FilterError`` on anything it cannot compile โ€” including + an empty object, which is far more likely to be a bug in the caller than a + request for "no filtering". + + An already-compiled filter passes straight through, so plumbing can call + this at every layer without recompiling. + """ + if filters is None: + return None + if isinstance(filters, CompiledFilter): + return filters + if not isinstance(filters, dict): + raise FilterError( + f"filters must be a JSON object, got {type(filters).__name__}. " + f"Supported fields: {_describe_support()}." + ) + if not filters: + raise FilterError( + "filters is empty. Omit the field entirely to search without a filter; " + "an empty filter object is refused so that a client bug cannot look " + "like an unfiltered search." + ) + + unknown = [k for k in filters if k not in _FIELD_BY_NAME] + if unknown: + raise FilterError( + f"Unsupported filter field(s): {', '.join(map(str, sorted(map(str, unknown))))}. " + f"Supported: {_describe_support()}." + ) + + predicates: List[str] = [] + for name, column, kind, allowed in _FIELDS: # canonical field order + if name not in filters: + continue + predicates.extend(_compile_field(name, column, kind, allowed, filters[name])) + + if not predicates: + raise FilterError("filters produced no predicates.") + + return CompiledFilter(where=" AND ".join(predicates), spec=dict(filters)) + + +def combine(where: Optional[str], compiled: Optional[CompiledFilter]) -> Optional[str]: + """AND an internally-generated where-clause with the caller's filter. + + Used where the pipeline already restricts a search on its own (the + cross-reference hop, the overview prefilter's restrict mode): the user's + filter must still apply, and it must apply as a *narrowing*, never as a + replacement. + """ + user = compiled.where if compiled is not None else None + parts = [p for p in (where, user) if p] + if not parts: + return None + if len(parts) == 1: + return parts[0] + return " AND ".join(f"({p})" for p in parts) diff --git a/rag_system/retrieval/query_transformer.py b/rag_system/retrieval/query_transformer.py index 77ab5165..47365bfd 100644 --- a/rag_system/retrieval/query_transformer.py +++ b/rag_system/retrieval/query_transformer.py @@ -7,7 +7,31 @@ def __init__(self, llm_client: OllamaClient, llm_model: str): self.llm_client = llm_client self.llm_model = llm_model - def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None) -> List[str]: + def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None, max_sub_queries: int = 10, resolve_only: bool = False) -> List[str]: + """Decompose *query* into standalone sub-queries. + + Two prompt variants, selected on whether conversation history exists: + + - `_decompose_single_turn` โ€” the FROZEN prompt every single-turn bench + number was measured against (arm C..K). It must stay byte-identical: + even cosmetic edits (smart quotes, added examples) were measured to + shift temp-0 decompositions on 5-25/120 gold queries and cost 2 + Sonnet-confirmed rfc rows (arm L, 2026-08-16). Any change here + re-triggers the full 5-bench gate. + - `_decompose_multi_turn` โ€” adds last_assistant_answer (the decomposer + cannot otherwise resolve pronouns whose antecedent the ASSISTANT + introduced; measured wrong-entity resolution on multiturn.jsonl + mt_07) plus the ellipsis rule 1b (mt_09). History is user-queries + only: interleaving answers made the 4b model substitute a previous + turn's question (m1 arm, 11/12). + """ + if chat_history: + return self._decompose_multi_turn(query, chat_history, max_sub_queries, + resolve_only=resolve_only) + return self._decompose_single_turn(query, max_sub_queries=max_sub_queries, + resolve_only=resolve_only) + + def _decompose_multi_turn(self, query: str, chat_history: List[Dict[str, Any]], max_sub_queries: int = 10, resolve_only: bool = False) -> List[str]: """Decompose *query* into standalone sub-queries. Parameters @@ -18,31 +42,45 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None Recent conversation turns (each item should contain at least the original user query under the key ``"query"``). Only the **last 5** turns are included to keep the prompt short. + max_sub_queries : int + Hard cap on the returned sub-queries (``query_decomposition.max_sub_queries``). """ - # ---- Limit history to last 5 user turns and extract the queries ---- + # ---- History for context resolution: the last 5 user queries, plus + # ONLY the last assistant answer as a separate field. Without any + # answer the decomposer cannot resolve pronouns whose antecedent was + # introduced by the assistant (user asks "who is the largest + # supplier?" -> answer names Acme -> "their lead time?" has no + # antecedent in user turns alone; measured resolving to the wrong + # entity on eval/goldset/multiturn.jsonl mt_07). Interleaving every + # answer into chat_history was measured worse: the 4b decomposer + # anchored on a previous turn and substituted its query (mt_09). history_snippets: List[str] = [] + last_assistant_answer = "" if chat_history: - # Keep only the last 5 turns recent_turns = chat_history[-5:] - # Extract user queries (fallback: full dict as string if key missing) for turn in recent_turns: - history_snippets.append(str(turn.get("query", turn))) + history_snippets.append( + str(turn.get("query", turn)) if isinstance(turn, dict) else str(turn)) + last_assistant_answer = " ".join( + str(recent_turns[-1].get("answer", "") or "").split()) if isinstance(recent_turns[-1], dict) else "" + if len(last_assistant_answer) > 300: + last_assistant_answer = last_assistant_answer[:300] + "..." # Serialize chat_history for the prompt (single string) chat_history_text = " | ".join(history_snippets) - # ---- Build the new SYSTEM prompt with added legacy examples ---- + # ---- Build the SYSTEM prompt ---- system_prompt = """ You are an expert at query decomposition for a Retrieval-Augmented Generation (RAG) system. Return one RFC-8259-compliant JSON object and nothing else. Schema: { -โ€œrequires_decompositionโ€: <bool>, -โ€œreasoningโ€: <string>, // โ‰ค 50 words -โ€œresolved_queryโ€: <string>, // query after context resolution -โ€œsub_queriesโ€: <string[]> // 1โ€“10 standalone items +"requires_decomposition": <bool>, +"reasoning": <string>, // โ‰ค 50 words +"resolved_query": <string>, // query after context resolution +"sub_queries": <string[]> // 1โ€“10 standalone items } Think step-by-step internally, but reveal only the concise reasoning. @@ -54,6 +92,7 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None You will receive: โ€ข query โ€“ the current user message โ€ข chat_history โ€“ the most recent user turns (may be empty) + โ€ข last_assistant_answer โ€“ what the assistant replied to the most recent turn (may be empty); use it only to resolve references that chat_history alone cannot If query contains pronouns, ellipsis, or shorthand that can be unambiguously linked to something in chat_history, rewrite it to a fully self-contained question and place the result in resolved_query. Otherwise, copy query into resolved_query unchanged. @@ -75,6 +114,7 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None Output rules 1. Use resolved_queryโ€”not the raw queryโ€”to decide on decomposition. + 1b. resolved_query must ask for the SAME fact as query. Never substitute an earlier question from chat_history: a follow-up like "And when was it approved?" asks a NEW fact even when it continues the previous topic. 2. If requires_decomposition is false, sub_queries must contain exactly resolved_query. 3. Otherwise, produce 2โ€“10 self-contained questions; avoid pronouns and shared context. @@ -97,6 +137,34 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None ] } +Context resolution (antecedent introduced by the assistant's answer) +chat_history: "Which company is the largest supplier?" +last_assistant_answer: "Acme Industrial is the largest supplier, providing 40 percent of volume." +query: "What is their lead time?" + +{ + "requires_decomposition": false, + "reasoning": "Pronoun resolved via the assistant's answer; single information need.", + "resolved_query": "What is the lead time of Acme Industrial?", + "sub_queries": [ + "What is the lead time of Acme Industrial?" + ] +} + +Ellipsis continuation asks a NEW fact (do not repeat the previous question) +chat_history: "What does a building permit cost? | When was the permit application submitted?" +last_assistant_answer: "The permit application was submitted on 3 March 2024." +query: "And when was it approved?" + +{ + "requires_decomposition": false, + "reasoning": "Follow-up introduces a new fact (approval date); ellipsis resolved without repeating the earlier question.", + "resolved_query": "When was the building permit application approved?", + "sub_queries": [ + "When was the building permit application approved?" + ] +} + Context resolution (single info need) chat_history: โ€œWhat is the email address of the computer vision consultants?โ€ query: โ€œWhat is the address?โ€ @@ -193,54 +261,233 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None } """ - # ---- Append legacy examples that already existed in the old prompt ---- - legacy_examples_header = """ + full_prompt = ( + system_prompt + + new_examples + + """ + โธป -Additional legacy examples +Now process + +Input payload: + +""" + json.dumps({"query": query, "chat_history": chat_history_text, + "last_assistant_answer": last_assistant_answer}, indent=2) + """ +""" + ) + + # ---- Call the LLM ---- + # Greedy decode: sampled decomposition made the SAME query split + # differently run-to-run, which both destabilizes answers and made + # every synthesis A/B compare partially-different row sets. + response = self.llm_client.generate_completion( + self.llm_model, full_prompt, format="json", + options={"temperature": 0}) + + response_text = response.get('response', '{}') + try: + # Handle potential markdown code blocks in the response + if response_text.strip().startswith("```json"): + response_text = response_text.strip()[7:-3].strip() + + data = json.loads(response_text) + + sub_queries = data.get('sub_queries') or [query] + reasoning = data.get('reasoning', 'No reasoning provided.') + + print(f"Query Decomposition Reasoning: {reasoning}") + + # Resolve-only mode (component ablation 2026-08-18: splitting + # single-turn queries measured +3-real WORSE than not splitting, + # while context resolution is what the multi-turn fix needs): + # keep the same LLM call, use only its resolved_query. + if resolve_only: + resolved = str(data.get('resolved_query') or '').strip() or query + print(f"Resolve-only: using resolved query '{resolved}'") + return [resolved] + + # Deduplicate while preserving order + sub_queries = list(dict.fromkeys(sub_queries)) + + return sub_queries[:max(1, int(max_sub_queries))] + except json.JSONDecodeError: + print(f"Failed to decode JSON from query decomposer: {response_text}") + return [query] + +# GraphQueryTranslator was removed on 2026-08-09 (roadmap item 2.5) together with +# GraphExtractor and GraphRetriever. Evidence: Documentation/research/ +# academic-evidence-2026.md ยง6 โ€” GraphRAG loses on single-hop, its multi-hop +# gains are contested, and it costs 41โ€“57x at indexing and up to ~377x in query +# tokens. Nothing in this repo ever armed it. + def _decompose_single_turn(self, query: str, max_sub_queries: int = 10, resolve_only: bool = False) -> List[str]: + """Decompose *query* into standalone sub-queries. + + Parameters + ---------- + query : str + The latest user message. + max_sub_queries : int + Hard cap on the returned sub-queries (``query_decomposition.max_sub_queries``). + """ + + # ---- Build the SYSTEM prompt ---- + system_prompt = """ +You are an expert at query decomposition for a Retrieval-Augmented Generation (RAG) system. + +Return one RFC-8259-compliant JSON object and nothing else. +Schema: +{ +โ€œrequires_decompositionโ€: <bool>, +โ€œreasoningโ€: <string>, // โ‰ค 50 words +โ€œresolved_queryโ€: <string>, // query after context resolution +โ€œsub_queriesโ€: <string[]> // 1โ€“10 standalone items +} + +Think step-by-step internally, but reveal only the concise reasoning. + +โธป + +Context Resolution (perform FIRST) + +You will receive: + โ€ข query โ€“ the current user message + โ€ข chat_history โ€“ the most recent user turns (may be empty) + +If query contains pronouns, ellipsis, or shorthand that can be unambiguously linked to something in chat_history, rewrite it to a fully self-contained question and place the result in resolved_query. +Otherwise, copy query into resolved_query unchanged. + +โธป + +When is decomposition REQUIRED? + โ€ข MULTI-PART questions joined by โ€œandโ€, โ€œorโ€, โ€œalsoโ€, list commas, etc. + โ€ข COMPARATIVE / SUPERLATIVE questions (two or more entities, e.g. โ€œbigger, better, fastestโ€). + โ€ข TEMPORAL / SEQUENTIAL questions (changes over time, event timelines). + โ€ข ENUMERATIONS (pros, cons, impacts). + โ€ข ENTITY-SET COMPARISONS (A, B, C revenueโ€ฆ). + +When is decomposition NOT REQUIRED? + โ€ข A single, factual information need. + โ€ข Ambiguous queries needing clarification rather than splitting. + +โธป + +Output rules + 1. Use resolved_queryโ€”not the raw queryโ€”to decide on decomposition. + 2. If requires_decomposition is false, sub_queries must contain exactly resolved_query. + 3. Otherwise, produce 2โ€“10 self-contained questions; avoid pronouns and shared context. + +โธป """ - legacy_examples_body = """ -**Example 1: Multi-Part Query** -Query: "What were the main findings of the aiconfig report and how do they compare to the results from the RAG paper?" -JSON Output: + # ---- Append NEW examples provided by the user ---- + new_examples = """ + +Normalise pronouns and references: turn โ€œthis paperโ€ into the explicit title if it can be inferred, otherwise leave as-is. +chat_history: โ€œWhat is the email address of the computer vision consultants?โ€ +query: โ€œWhat is their revenue?โ€ + { - "reasoning": "The query asks for two distinct pieces of information: the findings from one report and a comparison to another. This requires two separate retrieval steps.", + "requires_decomposition": false, + "reasoning": "Pronoun resolved; single information need.", + "resolved_query": "What is the revenue of the computer vision consultants?", "sub_queries": [ - "What were the main findings of the aiconfig report?", - "How do the findings of the aiconfig report compare to the results from the RAG paper?" + "What is the revenue of the computer vision consultants?" ] } -**Example 2: Simple Query** -Query: "Summarize the contributions of the DeepSeek-V3 paper." -JSON Output: +Context resolution (single info need) +chat_history: โ€œWhat is the email address of the computer vision consultants?โ€ +query: โ€œWhat is the address?โ€ + { - "reasoning": "This is a direct request for a summary of a single document and does not contain multiple parts.", + "requires_decomposition": false, + "reasoning": "Pronoun resolved; single information need.", + "resolved_query": "What is the physical address of the computer vision consultants?", "sub_queries": [ - "Summarize the contributions of the DeepSeek-V3 paper." + "What is the physical address of the computer vision consultants?" ] } -**Example 3: Comparative Query** -Query: "Did Microsoft or Google make more money last year?" -JSON Output: +Context resolution (single info need) +chat_history: โ€œComputeX has a revenue of 100M?โ€ +query: โ€œWho is the CEO?โ€ + { - "reasoning": "This is a comparative query that requires fetching the profit for each company before a comparison can be made.", + "requires_decomposition": false, + "reasoning": "entities normalization.", + "resolved_query": "who is the CEO of ComputeX", "sub_queries": [ - "How much profit did Microsoft make last year?", - "How much profit did Google make last year?" + "who is the CEO of ComputeX" ] } -**Example 4: Comparative Query with different phrasing** -Query: "Who has more siblings, Jamie or Sansa?" -JSON Output: +No unique antecedent โ†’ leave unresolved +chat_history: โ€œTell me about the paper.โ€ +query: โ€œWhat is the address?โ€ + { - "reasoning": "This comparative query needs the sibling count for both individuals to be answered.", + "requires_decomposition": false, + "reasoning": "Ambiguous reference; cannot resolve safely.", + "resolved_query": "What is the address?", + "sub_queries": ["What is the address?"] +} + +Temporal + Comparative +chat_history: "" +query: โ€œHow did Nvidiaโ€™s 2024 revenue compare with 2023?โ€ + +{ + "requires_decomposition": true, + "reasoning": "Needs revenue for two separate years before comparison.", + "resolved_query": "How did Nvidiaโ€™s 2024 revenue compare with 2023?", "sub_queries": [ - "How many siblings does Jamie have?", - "How many siblings does Sansa have?" + "What was Nvidiaโ€™s revenue in 2024?", + "What was Nvidiaโ€™s revenue in 2023?" + ] +} + +Enumeration (pros / cons / cost) +chat_history: "" +query: โ€œList the pros, cons, and estimated implementation cost of adopting a vector database.โ€ + +{ + "requires_decomposition": true, + "reasoning": "Three distinct information needs: pros, cons, cost.", + "resolved_query": "List the pros, cons, and estimated implementation cost of adopting a vector database.", + "sub_queries": [ + "What are the pros of adopting a vector database?", + "What are the cons of adopting a vector database?", + "What is the estimated implementation cost of adopting a vector database?" + ] +} + +Entity-set comparison (multiple companies) +chat_history: "" +query: โ€œHow did Nvidia, AMD, and Intel perform in Q2 2025 in terms of revenue?โ€ + +{ + "requires_decomposition": true, + "reasoning": "Need revenue for each of three entities before comparison.", + "resolved_query": "How did Nvidia, AMD, and Intel perform in Q2 2025 in terms of revenue?", + "sub_queries": [ + "What was Nvidia's revenue in Q2 2025?", + "What was AMD's revenue in Q2 2025?", + "What was Intel's revenue in Q2 2025?" + ] +} + +Multi-part question (limitations + mitigations) +chat_history: "" +query: โ€œWhat are the limitations of GPT-4o and what are the recommended mitigations?โ€ + +{ + "requires_decomposition": true, + "reasoning": "Two distinct pieces of information: limitations and mitigations.", + "resolved_query": "What are the limitations of GPT-4o and what are the recommended mitigations?", + "sub_queries": [ + "What are the known limitations of GPT-4o?", + "What are the recommended mitigations for the limitations of GPT-4o?" ] } """ @@ -248,8 +495,6 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None full_prompt = ( system_prompt + new_examples - # + legacy_examples_header - # + legacy_examples_body + """ โธป @@ -258,12 +503,17 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None Input payload: -""" + json.dumps({"query": query, "chat_history": chat_history_text}, indent=2) + """ +""" + json.dumps({"query": query, "chat_history": ""}, indent=2) + """ """ ) # ---- Call the LLM ---- - response = self.llm_client.generate_completion(self.llm_model, full_prompt, format="json") + # Greedy decode: sampled decomposition made the SAME query split + # differently run-to-run, which both destabilizes answers and made + # every synthesis A/B compare partially-different row sets. + response = self.llm_client.generate_completion( + self.llm_model, full_prompt, format="json", + options={"temperature": 0}) response_text = response.get('response', '{}') try: @@ -278,50 +528,25 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None print(f"Query Decomposition Reasoning: {reasoning}") - # Fallback: ensure at least the resolved_query if sub_queries empty - if not sub_queries: - sub_queries = [data.get('resolved_query', query)] + # Resolve-only mode (component ablation 2026-08-18: splitting + # single-turn queries measured +3-real WORSE than not splitting, + # while context resolution is what the multi-turn fix needs): + # keep the same LLM call, use only its resolved_query. + if resolve_only: + resolved = str(data.get('resolved_query') or '').strip() or query + print(f"Resolve-only: using resolved query '{resolved}'") + return [resolved] # Deduplicate while preserving order sub_queries = list(dict.fromkeys(sub_queries)) - # Enforce 10 sub-query limit per new requirements - return sub_queries[:10] + return sub_queries[:max(1, int(max_sub_queries))] except json.JSONDecodeError: print(f"Failed to decode JSON from query decomposer: {response_text}") return [query] -class HyDEGenerator: - def __init__(self, llm_client: OllamaClient, llm_model: str): - self.llm_client = llm_client - self.llm_model = llm_model - - def generate(self, query: str) -> str: - prompt = f"Generate a short, hypothetical document that answers the following question. The document should be dense with keywords and concepts related to the query.\n\nQuery: {query}\n\nHypothetical Document:" - response = self.llm_client.generate_completion(self.llm_model, prompt) - return response.get('response', '') - -class GraphQueryTranslator: - def __init__(self, llm_client: OllamaClient, llm_model: str): - self.llm_client = llm_client - self.llm_model = llm_model - - def _generate_translation_prompt(self, query: str) -> str: - return f""" -You are an expert query planner. Convert the user's question into a structured JSON query for a knowledge graph. -The JSON should contain a 'start_node' (the known entity in the query) and an 'edge_label' (the relationship being asked about). -The graph has nodes (entities) and directed edges (relationships). For example, (Tim Cook) -[IS_CEO_OF]-> (Apple). -Return ONLY the JSON object. - -User Question: "{query}" - -JSON Output: -""" - - def translate(self, query: str) -> Dict[str, Any]: - prompt = self._generate_translation_prompt(query) - response = self.llm_client.generate_completion(self.llm_model, prompt, format="json") - try: - return json.loads(response.get('response', '{}')) - except json.JSONDecodeError: - return {} \ No newline at end of file +# GraphQueryTranslator was removed on 2026-08-09 (roadmap item 2.5) together with +# GraphExtractor and GraphRetriever. Evidence: Documentation/research/ +# academic-evidence-2026.md ยง6 โ€” GraphRAG loses on single-hop, its multi-hop +# gains are contested, and it costs 41โ€“57x at indexing and up to ~377x in query +# tokens. Nothing in this repo ever armed it. \ No newline at end of file diff --git a/rag_system/retrieval/retrievers.py b/rag_system/retrieval/retrievers.py index 2b418d8d..53d4c6a3 100644 --- a/rag_system/retrieval/retrievers.py +++ b/rag_system/retrieval/retrievers.py @@ -1,200 +1,360 @@ -import lancedb -import pickle import json -from typing import List, Dict, Any -import numpy as np -import networkx as nx import os -from PIL import Image -from transformers import CLIPProcessor, CLIPModel -import torch +import urllib.request +from typing import Any, Dict, List, Optional, Tuple import logging -import pandas as pd import math import concurrent.futures from functools import lru_cache -from rag_system.indexing.embedders import LanceDBManager +from rag_system.indexing.embedders import ( + EmbedderMismatchError, + LanceDBManager, + assert_embedder_matches, + l2_normalize, + legacy_table_warning, + read_table_marker, +) from rag_system.indexing.representations import QwenEmbedder -from rag_system.indexing.multimodal import LocalVisionModel from rag_system.utils.logging_utils import log_retrieval_results -# BM25Retriever is no longer needed. -# class BM25Retriever: ... - -from fuzzywuzzy import process - -class GraphRetriever: - def __init__(self, graph_path: str): - self.graph = nx.read_gml(graph_path) - - def retrieve(self, query: str, k: int = 5, score_cutoff: int = 80) -> List[Dict[str, Any]]: - print(f"\n--- Performing Graph Retrieval for query: '{query}' ---") - - query_parts = query.split() - entities = [] - for part in query_parts: - match = process.extractOne(part, self.graph.nodes(), score_cutoff=score_cutoff) - if match and isinstance(match[0], str): - entities.append(match[0]) - - retrieved_docs = [] - for entity in set(entities): - for neighbor in self.graph.neighbors(entity): - retrieved_docs.append({ - 'chunk_id': f"graph_{entity}_{neighbor}", - 'text': f"Entity: {entity}, Neighbor: {neighbor}", - 'score': 1.0, - 'metadata': {'source': 'graph'} - }) - - print(f"Retrieved {len(retrieved_docs)} documents from the graph.") - return retrieved_docs[:k] +# Retrieval modes accepted by MultiVectorRetriever.retrieve(). +RETRIEVAL_MODES = ("hybrid", "vector_only", "fts_only") + +# Reciprocal-rank-fusion constant (Cormack et al. 2009); dampens the influence +# of the top ranks so a single leg cannot dominate the fused ordering. +_RRF_K = 60 + + +def _is_nan(value: Any) -> bool: + return isinstance(value, float) and math.isnan(value) + + +def _finite(value: Any) -> Optional[float]: + """Return *value* as a float, or None when it is missing/NaN/Inf.""" + if value is None or _is_nan(value): + return None + try: + num = float(value) + except (TypeError, ValueError): + return None + if math.isnan(num) or math.isinf(num): + return None + return num + + +# GraphRetriever was removed on 2026-08-09 (roadmap item 2.5). GraphRAG loses on +# single-hop retrieval, its multi-hop gains are contested, and it costs 41โ€“57x at +# indexing and up to ~377x in query tokens โ€” see Documentation/research/ +# academic-evidence-2026.md ยง6. It was also unreachable: no shipped profile ever +# set `graph_strategy`. + # region === MultiVectorRetriever === class MultiVectorRetriever: """ - Performs hybrid (vector + FTS) or vector-only retrieval. + Runs LanceDB full-text and/or vector search over a single table. """ - def __init__(self, db_manager: LanceDBManager, text_embedder: QwenEmbedder, vision_model: LocalVisionModel = None, *, fusion_config: Dict[str, Any] | None = None): + def __init__(self, db_manager: LanceDBManager, text_embedder: QwenEmbedder): self.db_manager = db_manager self.text_embedder = text_embedder - self.vision_model = vision_model - self.fusion_config = fusion_config or {"method": "linear", "bm25_weight": 0.5, "vec_weight": 0.5} - # Lightweight in-memory LRU cache for single-query embeddings (256 entries) + # Lightweight in-memory LRU cache for single-query embeddings (256 entries). + # The cache holds the raw model output; normalization is applied after the + # lookup because it is a property of the table being searched, not of the + # query (a legacy table must be searched with legacy vectors). @lru_cache(maxsize=256) def _embed_single(q: str): return self.text_embedder.create_embeddings([q])[0] self._embed_single = _embed_single - def retrieve(self, text_query: str, table_name: str, k: int, reranker=None) -> List[Dict[str, Any]]: + # Tables already reported as unmarked, so the warning is printed once each. + self._legacy_tables_warned: set = set() + + def _check_table_identity(self, tbl, table_name: str) -> bool: + """Verify the table's embedder against ours; return whether to normalize. + + Raises ``EmbedderMismatchError`` when the table was written by a + different embedding model โ€” a same-width swap (harrier-oss-v1-0.6b vs + Qwen3-Embedding-0.6B, both 1024-dim) is otherwise undetectable and would + silently return nonsense. """ - Performs a search on a single LanceDB table. - If a reranker is provided, it performs a hybrid search. - Otherwise, it performs a standard vector search. + marker = read_table_marker( + tbl, getattr(self.db_manager, "db_path", None), table_name + ) + if marker is None: + if table_name not in self._legacy_tables_warned: + self._legacy_tables_warned.add(table_name) + print(legacy_table_warning(table_name)) + return False + configured = getattr(self.text_embedder, "model_name", None) + if configured: + assert_embedder_matches(table_name, marker, configured) + return bool(marker["normalized"]) + + @staticmethod + def _dedup_key(row: Dict[str, Any]) -> Tuple[str, Any]: + for field in ("chunk_id", "_rowid"): + value = row.get(field) + if value is not None and not _is_nan(value): + return (field, value) + return ("text", row.get("text")) + + @staticmethod + def _mv_sidecar_search(text_query: str, table_name: str, k: int, + where: Optional[str]): + """EXPERIMENTAL: multi-vector (late-interaction) replacement for the + dense leg, gated on the ``MV_RETRIEVAL_ENDPOINT`` env var. + + The sidecar (an isolated SentenceTransformers-v6 process โ€” ST6 needs + transformers 5.x / torch 2.5+, incompatible with this repo's pinned + MPS stack) holds pre-encoded token embeddings for the table and + returns the top-k rows by MaxSim. Returns None when the env var is + unset (normal operation: not a single extra call is made). When it IS + set, any failure raises rather than silently degrading to the dense + leg โ€” a benchmark arm that quietly reverted to the control would + produce numbers indistinguishable from "no effect". + + Metadata prefilters are not implemented on this path for the same + reason: raising beats silently ignoring the filter. + """ + endpoint = os.environ.get("MV_RETRIEVAL_ENDPOINT") + if not endpoint: + return None + if where: + raise RuntimeError( + "MV_RETRIEVAL_ENDPOINT is set but this query carries a " + "metadata prefilter, which the multi-vector sidecar does not " + "implement.") + payload = json.dumps({"table": table_name, "query": text_query, + "k": k}).encode() + req = urllib.request.Request( + endpoint.rstrip("/") + "/search", data=payload, + headers={"Content-Type": "application/json"}) + with urllib.request.urlopen(req, timeout=120) as resp: + data = json.loads(resp.read()) + if "rows" not in data: + raise RuntimeError(f"MV sidecar error: {data}") + import pandas as pd + return pd.DataFrame(data["rows"]) + + def retrieve(self, text_query: str, table_name: str, k: int, search_type: str = "hybrid", + *, where: Optional[str] = None) -> List[Dict[str, Any]]: """ - print(f"\n--- Performing Retrieval for query: '{text_query}' on table '{table_name}' ---") - + Retrieves up to *k* chunks from *table_name*. + + `search_type` selects which LanceDB query legs run: "hybrid" (full-text + + vector, fused with reciprocal rank fusion), "vector_only", or + "fts_only". Unknown values fall back to "hybrid". + + `where` is a LanceDB SQL predicate applied as a **prefilter to every + leg** (roadmap item 4.4). It is keyword-only and defaults to None, in + which case not a single extra call is made and the behaviour is exactly + what it was before filters existed. It must come from + ``rag_system.retrieval.filters.compile_filters`` โ€” nothing here + validates or escapes it, and nothing here will repair it. + + Prefiltering (rather than filtering the k results afterwards) is the + point: a post-filter would return "up to k, minus whatever the filter + removed", so a filter for a rare document would return an empty list + while the document sat in the table. Both legs are filtered, so a hybrid + query cannot leak an excluded chunk in through the BM25 side. + """ + mode = (search_type or "hybrid").lower() + if mode not in RETRIEVAL_MODES: + print(f"โš ๏ธ Unknown retrieval mode '{search_type}'; falling back to 'hybrid'.") + mode = "hybrid" + + print(f"\n--- Performing {mode} retrieval for query: '{text_query}' on table '{table_name}' ---") + if where: + print(f"๐Ÿ”Ž Metadata filter (prefilter, both legs): {where}") + try: if table_name is None: table_name = "default_text_table" tbl = self.db_manager.get_table(table_name) - - # Create / fetch cached text embedding for the query - text_query_embedding = self._embed_single(text_query) - - logger = logging.getLogger(__name__) - - # Always perform hybrid lexical + vector search - logger.debug( - "Running hybrid search on table '%s' (k=%s, have_reranker=%s)", - table_name, - k, - bool(reranker), - ) + # Vectors are stored L2-normalized on v4+ tables, so LanceDB's default + # L2 ordering equals the cosine ordering both model cards specify. The + # query vector has to be normalized the same way โ€” and must *not* be + # for a legacy table, whose documents are unnormalized. + normalize_query = self._check_table_identity(tbl, table_name) - if reranker: - logger.debug("Hybrid + reranker path not yet implemented with manual fusion; proceeding without extra reranker.") + logger = logging.getLogger(__name__) + logger.debug("Running %s search on table '%s' (k=%s)", mode, table_name, k) - # Manual two-leg hybrid: take half from each modality - fts_k = k // 2 - vec_k = k - fts_k + def _filtered(query_builder): + """Apply the metadata prefilter, when there is one.""" + if where: + return query_builder.where(where, prefilter=True) + return query_builder - # Run FTS and vector search in parallel to cut latency def _run_fts(): + # LanceDB's FTS parser reads double quotes as phrase syntax and + # raises on queries it cannot parse โ€” the decomposer emits + # quoted sub-queries like `"Extra Fees & Costs" charges`, which + # used to kill the whole retrieve() (hybrid included) and return + # nothing. Quotes carry no ranking signal we rely on, so strip + # them and search the plain terms. + fts_query = text_query.replace('"', " ").strip() or text_query # Very short queries often underperform โ†’ add fuzzy wildcard - fts_query = text_query - if len(text_query.split()) == 1: - fts_query = f"{text_query}* OR {text_query}~" + if len(fts_query.split()) == 1: + fts_query = f"{fts_query}* OR {fts_query}~" return ( - tbl.search(query=fts_query, query_type="fts") - .limit(fts_k) - .to_df() - ) + _filtered(tbl.search(query=fts_query, query_type="fts")) + .limit(k) + .to_pandas() + ) + + # EXPERIMENTAL (see _mv_sidecar_search): with MV_RRF_LEG set, the + # multi-vector sidecar runs as a THIRD RRF leg next to FTS and + # dense; without it, it REPLACES the dense leg. MV_UNION implies + # the third leg but skips the top-k cut after fusion: the FULL + # candidate union goes downstream so the reranker arbitrates + # between legs instead of reciprocal-rank fusion โ€” a disagreeing + # leg can then only add candidates, never push another leg's + # find out of the pool. + mv_union = bool(os.environ.get("MV_UNION")) + mv_as_third_leg = mv_union or bool(os.environ.get("MV_RRF_LEG")) def _run_vec(): - if vec_k == 0: - return None + if not mv_as_third_leg: + mv_df = self._mv_sidecar_search(text_query, table_name, k, where) + if mv_df is not None: + return mv_df + vector = self._embed_single(text_query) + if normalize_query: + vector = l2_normalize(vector) return ( - tbl.search(text_query_embedding) - .limit(vec_k * 2) # fetch extra to allow for dedup - .to_df() + _filtered(tbl.search(vector)) + .limit(k) + .to_pandas() ) - with concurrent.futures.ThreadPoolExecutor(max_workers=2) as executor: - fts_future = executor.submit(_run_fts) - vec_future = executor.submit(_run_vec) - fts_df = fts_future.result() - vec_df = vec_future.result() - - if vec_df is not None: - combined = pd.concat([fts_df, vec_df]) + fts_df = None + vec_df = None + if mode == "fts_only": + fts_df = _run_fts() + elif mode == "vector_only": + vec_df = _run_vec() else: - combined = fts_df + with concurrent.futures.ThreadPoolExecutor(max_workers=2) as executor: + fts_future = executor.submit(_run_fts) + vec_future = executor.submit(_run_vec) + # In hybrid mode a failed FTS leg degrades to dense-only + # instead of failing the whole retrieval โ€” the two legs are + # redundant by design, and "one leg down" must not become + # "0 results". (fts_only mode still propagates, because + # there the caller asked for exactly that leg.) + try: + fts_df = fts_future.result() + except Exception as fts_err: + print(f"โš ๏ธ FTS leg failed ({fts_err}); continuing dense-only.") + logger.warning("FTS leg failed on table '%s': %s", table_name, fts_err) + fts_df = None + vec_df = vec_future.result() + + def _records(df) -> List[Dict[str, Any]]: + if df is None or len(df) == 0: + return [] + return df.to_dict("records") - # Remove duplicates preserving first occurrence, then trim to k - dedup_subset = ["_rowid"] if "_rowid" in combined.columns else (["chunk_id"] if "chunk_id" in combined.columns else None) - if dedup_subset: - combined = combined.drop_duplicates(subset=dedup_subset, keep="first") - combined = combined.head(k) + fts_rows = _records(fts_df) + vec_rows = _records(vec_df) + + mv_rows: List[Dict[str, Any]] = [] + if mv_as_third_leg and mode != "fts_only": + mv_df = self._mv_sidecar_search(text_query, table_name, k, where) + if mv_df is not None: + mv_rows = _records(mv_df) + + # Reciprocal rank fusion: each leg contributes 1/(_RRF_K + rank). + # A single-leg run keeps its native ordering because RRF is + # monotonically decreasing in rank. + fused: Dict[Tuple[str, Any], Dict[str, Any]] = {} + for rank, row in enumerate(fts_rows, start=1): + entry = fused.setdefault(self._dedup_key(row), {"row": row, "rrf": 0.0, "bm25": None, "distance": None}) + entry["rrf"] += 1.0 / (_RRF_K + rank) + # LanceDB FTS scores come back as `_score` (lancedb 0.36), not + # `score` โ€” reading the latter left bm25 permanently None. + entry["bm25"] = _finite(row.get("_score")) + for rank, row in enumerate(vec_rows, start=1): + entry = fused.setdefault(self._dedup_key(row), {"row": row, "rrf": 0.0, "bm25": None, "distance": None}) + entry["rrf"] += 1.0 / (_RRF_K + rank) + entry["distance"] = _finite(row.get("_distance")) + for rank, row in enumerate(mv_rows, start=1): + entry = fused.setdefault(self._dedup_key(row), {"row": row, "rrf": 0.0, "bm25": None, "distance": None}) + entry["rrf"] += 1.0 / (_RRF_K + rank) + + ordered = sorted(fused.values(), key=lambda e: e["rrf"], reverse=True) + if not mv_union: + ordered = ordered[:k] - results_df = combined logger.debug( - "Hybrid (fts=%s, vec=%s) โ†’ %s unique chunks", - len(fts_df), - 0 if vec_df is None else len(vec_df), - len(results_df), + "%s search (fts=%s, vec=%s) โ†’ %s unique chunks", + mode, + len(fts_rows), + len(vec_rows), + len(ordered), ) - + retrieved_docs = [] - for _, row in results_df.iterrows(): - metadata = json.loads(row.get('metadata', '{}')) + for entry in ordered: + row = entry["row"] + raw_metadata = row.get('metadata') + if isinstance(raw_metadata, dict): + metadata = dict(raw_metadata) + else: + try: + metadata = json.loads(raw_metadata or '{}') + except (TypeError, ValueError): + metadata = {} # Add top-level fields back into metadata for consistency if they don't exist metadata.setdefault('document_id', row.get('document_id')) metadata.setdefault('chunk_index', row.get('chunk_index')) - - # Determine score (vector distance or FTS). Replace NaN with 0.0 - raw_score = row.get('_distance') if '_distance' in row else row.get('score') - try: - if raw_score is None or (isinstance(raw_score, float) and math.isnan(raw_score)): - raw_score = 0.0 - except Exception: - raw_score = 0.0 - - combined_score = raw_score - # Optional linear-weight fusion if both FTS & vector scores exist - if '_distance' in row and 'score' in row: - try: - bm25 = row.get('score', 0.0) - vec_sim = 1.0 / (1.0 + row.get('_distance', 1.0)) # convert distance to similarity - w_bm25 = float(self.fusion_config.get('bm25_weight', 0.5)) - w_vec = float(self.fusion_config.get('vec_weight', 0.5)) - combined_score = w_bm25 * bm25 + w_vec * vec_sim - except Exception: - pass - - retrieved_docs.append({ + + # A single score field per mode, always "higher is better". + if mode == "fts_only": + score = entry["bm25"] if entry["bm25"] is not None else 0.0 + elif mode == "vector_only": + distance = entry["distance"] + score = 1.0 / (1.0 + distance) if distance is not None else 0.0 + else: + score = entry["rrf"] + + doc = { 'chunk_id': row.get('chunk_id'), - 'text': metadata.get('original_text', row.get('text')), - 'score': combined_score, - 'bm25': row.get('score'), - '_distance': row.get('_distance'), + 'text': metadata.get('original_text') or row.get('text') or '', + 'score': score, 'document_id': row.get('document_id'), 'chunk_index': row.get('chunk_index'), 'metadata': metadata - }) + } + # Only carry the per-leg raw scores when that leg actually hit, + # so downstream sorting never sees a None. + if entry["bm25"] is not None: + doc['bm25'] = entry["bm25"] + if entry["distance"] is not None: + doc['_distance'] = entry["distance"] + retrieved_docs.append(doc) - logger.debug("Hybrid search returned %s results", len(retrieved_docs)) log_retrieval_results(retrieved_docs, k) print(f"Retrieved {len(retrieved_docs)} documents.") return retrieved_docs - + + except EmbedderMismatchError: + # Never degrade an embedder mismatch into "0 results" โ€” the whole + # point of the guard is that the user has to see it. + raise except Exception as e: + if where: + # A filtered search that fails must not look like a filtered + # search that matched nothing. The caller asked to be restricted + # to part of the corpus; "0 results" and "the restriction did + # not run" are different answers and only one of them is safe. + print(f"โŒ Filtered search failed on table '{table_name}' " + f"(where: {where}): {e}") + raise print(f"Could not search table '{table_name}': {e}") return [] # endregion - -if __name__ == '__main__': - print("retrievers.py updated for LanceDB FTS Hybrid Search.") diff --git a/rag_system/utils/batch_processor.py b/rag_system/utils/batch_processor.py index 25d1eaeb..6f108b7f 100644 --- a/rag_system/utils/batch_processor.py +++ b/rag_system/utils/batch_processor.py @@ -1,6 +1,6 @@ import time import logging -from typing import List, Dict, Any, Callable, Optional, Iterator +from typing import List, Dict, Any, Callable from contextlib import contextmanager import gc @@ -115,7 +115,14 @@ def process_in_batches( tracker.update(len(batch)) except Exception as e: - logger.error(f"Error in batch {batch_num}: {e}") + # Loud, not silent: a skipped batch means its items are + # MISSING from the results (for enrichment, those chunks + # get indexed without context; for embeddings they surface + # later as a length-mismatch ValueError). + logger.warning( + f"โš ๏ธ {operation_name}: batch {batch_num}/{total_batches} FAILED โ€” " + f"{len(batch)} item(s) dropped from the results. Error: {e}" + ) tracker.update(len(batch), errors=len(batch)) # Continue processing other batches continue @@ -126,76 +133,8 @@ def process_in_batches( tracker.finish() return results - - def batch_iterator(self, items: List[Any]) -> Iterator[List[Any]]: - """Generate batches as an iterator for memory-efficient processing""" - for i in range(0, len(items), self.batch_size): - yield items[i:i + self.batch_size] - -class StreamingProcessor: - """Process items one at a time with minimal memory usage""" - - def __init__(self, enable_gc_interval: int = 100): - self.enable_gc_interval = enable_gc_interval - - def process_streaming( - self, - items: List[Any], - process_func: Callable, - operation_name: str = "Streaming Processing", - **kwargs - ) -> List[Any]: - """ - Process items one at a time with minimal memory footprint - - Args: - items: List of items to process - process_func: Function to process each item - operation_name: Name for progress reporting - **kwargs: Additional arguments passed to process_func - - Returns: - List of results - """ - if not items: - logger.info(f"{operation_name}: No items to process") - return [] - - tracker = ProgressTracker(len(items), operation_name) - results = [] - - logger.info(f"Starting {operation_name} for {len(items)} items (streaming)") - - with timer(f"{operation_name} (streaming)"): - for i, item in enumerate(items): - try: - result = process_func(item, **kwargs) - results.append(result) - tracker.update(1) - - except Exception as e: - logger.error(f"Error processing item {i}: {e}") - tracker.update(1, errors=1) - continue - - # Periodic garbage collection - if self.enable_gc_interval and (i + 1) % self.enable_gc_interval == 0: - gc.collect() - - tracker.finish() - return results # Utility functions for common batch operations -def batch_chunks_by_document(chunks: List[Dict[str, Any]]) -> Dict[str, List[Dict[str, Any]]]: - """Group chunks by document_id for document-level batch processing""" - document_batches = {} - for chunk in chunks: - doc_id = chunk.get('metadata', {}).get('document_id', 'unknown') - if doc_id not in document_batches: - document_batches[doc_id] = [] - document_batches[doc_id].append(chunk) - return document_batches - def estimate_memory_usage(chunks: List[Dict[str, Any]]) -> float: """Estimate memory usage of chunks in MB""" if not chunks: diff --git a/rag_system/utils/logging_utils.py b/rag_system/utils/logging_utils.py index 88336584..e4bb1406 100644 --- a/rag_system/utils/logging_utils.py +++ b/rag_system/utils/logging_utils.py @@ -12,17 +12,6 @@ ) -def log_query(query: str, sub_queries: List[str] | None = None) -> None: - """Emit a nicely-formatted block describing the incoming query and any - decomposition.""" - border = "=" * 60 - logger.info("\n%s\nUSER QUERY: %s", border, query) - if sub_queries: - for i, q in enumerate(sub_queries, 1): - logger.info(" sub-%d โ†’ %s", i, q) - logger.info("%s", border) - - def log_retrieval_results(results: List[Dict], k: int) -> None: """Show chunk_id, truncated text and score for the first *k* rows.""" if not results: diff --git a/rag_system/utils/ollama_client.py b/rag_system/utils/ollama_client.py index ea979d5c..7bd7b50d 100644 --- a/rag_system/utils/ollama_client.py +++ b/rag_system/utils/ollama_client.py @@ -1,11 +1,213 @@ import requests import json -from typing import List, Dict, Any +import os +from typing import List, Dict, Any, Optional import base64 +import contextlib +import contextvars +import threading from io import BytesIO from PIL import Image import httpx, asyncio +# --------------------------------------------------------------------------- +# Context-window sizing (fixes silent front-truncation) +# --------------------------------------------------------------------------- +# Ollama's server-side window (OLLAMA_CONTEXT_LENGTH, default 4096) is split +# across parallel slots, and a request that exceeds its slot is FRONT-truncated +# with no error โ€” measured on this machine as an effective 8194-token ceiling +# that dropped the top-ranked context chunks from synthesis prompts. The fix is +# to size the window per request: `options.num_ctx` overrides the server value. +# +# Sizing is bucketed (8k/16k/32k) rather than exact, and per-model MONOTONIC: +# once a model has been granted a window it never shrinks back. Changing +# num_ctx between calls forces Ollama to reallocate the KV cache (measured: +# the smoke suite went 363s โ†’ 924s when the agent's small triage calls and +# large synthesis calls alternated buckets; a fixed window ran it in 259s), so +# the window may only ratchet upward โ€” at most two transitions per model per +# process. The estimate deliberately overshoots (~3 chars/token vs the ~4 +# typical for English) because undershooting reintroduces the truncation this +# exists to prevent. +# +# Env knobs: OLLAMA_NUM_CTX pins an exact value (skips estimation); +# OLLAMA_NUM_CTX_MAX caps the bucket (default 32768 โ€” raise it on machines +# with memory to spare, at the cost of a larger KV cache). + +_NUM_CTX_BUCKETS = (8192, 16384, 32768) +_OUTPUT_HEADROOM_TOKENS = 2048 +_MODEL_CTX_HIGH_WATER: Dict[str, int] = {} +_MODEL_CTX_LOCK = threading.Lock() + + +def _num_ctx_for(char_count: int) -> int: + pinned = os.getenv("OLLAMA_NUM_CTX") + if pinned: + try: + return max(2048, int(pinned)) + except ValueError: + pass + try: + max_ctx = int(os.getenv("OLLAMA_NUM_CTX_MAX", "32768")) + except ValueError: + max_ctx = 32768 + estimated = char_count // 3 + _OUTPUT_HEADROOM_TOKENS + for bucket in _NUM_CTX_BUCKETS: + if estimated <= bucket <= max_ctx: + return bucket + return max_ctx + + +def _apply_num_ctx(payload: Dict[str, Any]) -> None: + """Set options.num_ctx from the prompt size, preserving caller options. + + Per-model monotonic: the requested window never shrinks below the largest + one this process has already used for the model (see module comment). + """ + options = payload.setdefault("options", {}) + if "num_ctx" in options: + return + needed = _num_ctx_for(len(payload.get("prompt") or "")) + model = str(payload.get("model") or "") + with _MODEL_CTX_LOCK: + granted = max(needed, _MODEL_CTX_HIGH_WATER.get(model, 0)) + _MODEL_CTX_HIGH_WATER[model] = granted + options["num_ctx"] = granted + + +def _warn_if_truncated(payload: Dict[str, Any], final_response: Dict[str, Any]) -> None: + """Detect a filled context window โ€” the signature of front-truncation. + + Best-effort diagnostics only; never raises. + """ + try: + num_ctx = int(payload.get("options", {}).get("num_ctx") or 0) + used = int(final_response.get("prompt_eval_count") or 0) + prompt_chars = len(payload.get("prompt") or "") + if num_ctx and used >= num_ctx - 16: + print( + f"โš ๏ธ Ollama likely FRONT-TRUNCATED this prompt: " + f"prompt_eval_count={used} filled num_ctx={num_ctx} " + f"(model={payload.get('model')}). Raise OLLAMA_NUM_CTX_MAX or " + f"shrink the prompt โ€” the beginning of the prompt was dropped." + ) + elif used and prompt_chars and used < prompt_chars // 6: + # Slot-proof check: the server divides num_ctx across its parallel + # slots, so a request can be truncated far BELOW the window we + # asked for and the filled-window check above stays silent + # (measured: 94k-token prompt, num_ctx 32768, served 16386). + # English text runs ~3.5โ€“4.5 chars/token, so evaluating fewer than + # chars/6 tokens means part of the prompt never reached the model. + print( + f"โš ๏ธ Ollama FRONT-TRUNCATED this prompt below the requested window: " + f"prompt_eval_count={used} for a {prompt_chars}-char prompt " + f"(num_ctx={num_ctx}, model={payload.get('model')}). The served " + f"per-request window is num_ctx divided by the server's parallel " + f"slots โ€” set OLLAMA_NUM_PARALLEL=1 or shrink the prompt." + ) + except Exception: + pass + +# --------------------------------------------------------------------------- +# Per-query token tracking (roadmap item 4.5) +# --------------------------------------------------------------------------- +# Ollama returns ``prompt_eval_count`` (input tokens) and ``eval_count`` (output +# tokens) on the final object of every /api/generate response, streaming or not. +# They are free โ€” we already parse that object โ€” so the only work is routing +# them somewhere useful. +# +# The routing is a ``ContextVar`` rather than a client attribute because the one +# ``OllamaClient`` instance is shared by the agent, the retrieval pipeline, the +# verifier and the decomposer. A context variable attributes each call to +# whichever request is on the stack, and `asyncio.to_thread` / `await` propagate +# it for free. The one place that does NOT propagate it is a raw +# ``ThreadPoolExecutor.submit`` โ€” the agent's parallel sub-query fan-out โ€” which +# copies the context explicitly (see ``rag_system/agent/loop.py``). +# +# Nothing here can fail a request: every record path is best-effort. + +_CURRENT_TRACKER: contextvars.ContextVar[Optional["TokenUsageTracker"]] = contextvars.ContextVar( + "localgpt_token_tracker", default=None +) +_CURRENT_STAGE: contextvars.ContextVar[str] = contextvars.ContextVar( + "localgpt_token_stage", default="other" +) + + +class TokenUsageTracker: + """Aggregates LLM token counts for one user query, bucketed by stage. + + Stages are the agent's own pipeline phases (``triage``, ``decomposition``, + ``synthesis``, ``verification``, โ€ฆ). A bucket only appears once a call has + been attributed to it, so an absent key means "that stage made no LLM call", + not "zero tokens". + """ + + def __init__(self) -> None: + self._lock = threading.Lock() + self._by_stage: Dict[str, Dict[str, int]] = {} + + def record(self, stage: str, prompt_tokens: int, output_tokens: int) -> None: + with self._lock: + bucket = self._by_stage.setdefault( + stage, {"prompt_tokens": 0, "output_tokens": 0, "calls": 0} + ) + bucket["prompt_tokens"] += int(prompt_tokens or 0) + bucket["output_tokens"] += int(output_tokens or 0) + bucket["calls"] += 1 + + def as_dict(self) -> Dict[str, Any]: + with self._lock: + by_stage = {k: dict(v) for k, v in self._by_stage.items()} + total = { + "prompt_tokens": sum(b["prompt_tokens"] for b in by_stage.values()), + "output_tokens": sum(b["output_tokens"] for b in by_stage.values()), + "calls": sum(b["calls"] for b in by_stage.values()), + } + total["total_tokens"] = total["prompt_tokens"] + total["output_tokens"] + return {"by_stage": by_stage, "total": total} + + +@contextlib.contextmanager +def track_token_usage(tracker: Optional[TokenUsageTracker]): + """Bind *tracker* as the sink for LLM token counts inside this block.""" + token = _CURRENT_TRACKER.set(tracker) + try: + yield tracker + finally: + _CURRENT_TRACKER.reset(token) + + +@contextlib.contextmanager +def token_stage(name: str): + """Attribute every LLM call made inside this block to stage *name*.""" + token = _CURRENT_STAGE.set(name) + try: + yield + finally: + _CURRENT_STAGE.reset(token) + + +def record_llm_usage(payload: Dict[str, Any]) -> None: + """Record one completed LLM call from a raw Ollama-shaped response object. + + A no-op when no tracker is bound, or when the backend reports neither count + (watsonx). Never raises. + """ + tracker = _CURRENT_TRACKER.get() + if tracker is None or not isinstance(payload, dict): + return + if "prompt_eval_count" not in payload and "eval_count" not in payload: + return + try: + tracker.record( + _CURRENT_STAGE.get(), + payload.get("prompt_eval_count") or 0, + payload.get("eval_count") or 0, + ) + except Exception: + pass + + class OllamaClient: """ An enhanced client for Ollama that now handles image data for VLM models. @@ -21,18 +223,6 @@ def _image_to_base64(self, image: Image.Image) -> str: image.save(buffered, format="PNG") return base64.b64encode(buffered.getvalue()).decode('utf-8') - def generate_embedding(self, model: str, text: str) -> List[float]: - try: - response = requests.post( - f"{self.api_url}/embeddings", - json={"model": model, "prompt": text} - ) - response.raise_for_status() - return response.json().get("embedding", []) - except requests.exceptions.RequestException as e: - print(f"Error generating embedding: {e}") - return [] - def generate_completion( self, model: str, @@ -41,6 +231,8 @@ def generate_completion( format: str = "", images: List[Image.Image] | None = None, enable_thinking: bool | None = None, + options: Optional[Dict[str, Any]] = None, + timeout: int = 60, ) -> Dict[str, Any]: """ Generates a completion, now with optional support for images. @@ -51,6 +243,11 @@ def generate_completion( format: The format for the response, e.g., "json". images: A list of Pillow Image objects to send to the VLM. enable_thinking: Optional flag to disable chain-of-thought for Qwen models. + options: Ollama options dict (e.g. {"temperature": 0}); an explicit + num_ctx here wins over the automatic sizing. + timeout: Seconds before the HTTP call gives up (same default the + async variant uses) โ€” a hung Ollama must not stall index + builds, which call this once per chunk. """ try: payload = { @@ -60,21 +257,35 @@ def generate_completion( } if format: payload["format"] = format + if options: + payload["options"] = dict(options) if images: payload["images"] = [self._image_to_base64(img) for img in images] - # Optional: disable thinking mode for Qwen3 / DeepSeek models + # Thinking models put JSON into the `thinking` field and leave + # `response` empty when format=json, so default thinking off there. + # `think` is the top-level knob /api/generate actually honors + # (chat_template_kwargs is silently ignored by the generate API). + if enable_thinking is None and format == "json": + enable_thinking = False if enable_thinking is not None: - payload["chat_template_kwargs"] = {"enable_thinking": enable_thinking} + payload["think"] = enable_thinking + + _apply_num_ctx(payload) response = requests.post( f"{self.api_url}/generate", - json=payload + json=payload, + timeout=timeout ) response.raise_for_status() response_lines = response.text.strip().split('\n') final_response = json.loads(response_lines[-1]) + # roadmap 4.5: `prompt_eval_count` / `eval_count` ride along on this + # object already; hand them to the per-query tracker if one is bound. + record_llm_usage(final_response) + _warn_if_truncated(payload, final_response) return final_response except requests.exceptions.RequestException as e: @@ -94,24 +305,45 @@ async def generate_completion_async( images: List[Image.Image] | None = None, enable_thinking: bool | None = None, timeout: int = 60, + options: Optional[Dict[str, Any]] = None, ) -> Dict[str, Any]: - """Asynchronous version of generate_completion using httpx.""" + """Asynchronous version of generate_completion using httpx. + + *options* is an Ollama options dict (e.g. {"temperature": 0}); an + explicit num_ctx here wins over the automatic sizing. + """ payload = {"model": model, "prompt": prompt, "stream": False} if format: payload["format"] = format + if options: + # Caller-supplied sampling options (e.g. temperature). Merged + # before _apply_num_ctx so an explicit num_ctx here wins. + payload["options"] = dict(options) if images: payload["images"] = [self._image_to_base64(img) for img in images] + if enable_thinking is None and format == "json": + enable_thinking = False if enable_thinking is not None: - payload["chat_template_kwargs"] = {"enable_thinking": enable_thinking} + payload["think"] = enable_thinking + + _apply_num_ctx(payload) try: async with httpx.AsyncClient(timeout=timeout) as client: resp = await client.post(f"{self.api_url}/generate", json=payload) resp.raise_for_status() - return json.loads(resp.text.strip().split("\n")[-1]) - except (httpx.HTTPError, asyncio.CancelledError) as e: + final_response = json.loads(resp.text.strip().split("\n")[-1]) + record_llm_usage(final_response) + _warn_if_truncated(payload, final_response) + return final_response + except asyncio.CancelledError: + # Cancellation is control flow, not an LLM failure โ€” never swallow + # it into the {} fail-open below (that fail-open is documented + # behaviour for httpx errors only). + raise + except httpx.HTTPError as e: print(f"Async Ollama completion error: {e}") return {} @@ -125,6 +357,8 @@ def stream_completion( *, images: List[Image.Image] | None = None, enable_thinking: bool | None = None, + stats: Optional[Dict[str, Any]] = None, + options: Optional[Dict[str, Any]] = None, ): """Generator that yields partial *response* strings as they arrive. @@ -132,12 +366,24 @@ def stream_completion( for tok in client.stream_completion("qwen2", "Hello"): print(tok, end="", flush=True) + + A streaming generator cannot return a value, so pass a dict as *stats* + to receive the final Ollama object (``prompt_eval_count``, + ``eval_count``, timings) once the stream completes. Token counts are + also handed to the per-query tracker automatically (roadmap 4.5), which + is what the agent uses โ€” *stats* exists for direct callers. """ payload: Dict[str, Any] = {"model": model, "prompt": prompt, "stream": True} if images: payload["images"] = [self._image_to_base64(img) for img in images] if enable_thinking is not None: - payload["chat_template_kwargs"] = {"enable_thinking": enable_thinking} + payload["think"] = enable_thinking + if options: + # Caller-supplied sampling options (e.g. temperature). Merged + # before _apply_num_ctx so an explicit num_ctx here wins. + payload["options"] = dict(options) + + _apply_num_ctx(payload) with requests.post(f"{self.api_url}/generate", json=payload, stream=True) as resp: resp.raise_for_status() @@ -154,28 +400,10 @@ def stream_completion( if chunk: yield chunk if data.get("done"): + # The final object carries the token counts for the whole + # stream (roadmap 4.5). + if stats is not None: + stats.update(data) + record_llm_usage(data) + _warn_if_truncated(payload, data) break - -if __name__ == '__main__': - # This test now requires a VLM model like 'llava' or 'qwen-vl' to be pulled. - print("Ollama client updated for multimodal (VLM) support.") - try: - client = OllamaClient() - # Create a dummy black image for testing - dummy_image = Image.new('RGB', (100, 100), 'black') - - # Test VLM completion - vlm_response = client.generate_completion( - model="llava", # Make sure you have run 'ollama pull llava' - prompt="What color is this image?", - images=[dummy_image] - ) - - if vlm_response and 'response' in vlm_response: - print("\n--- VLM Test Response ---") - print(vlm_response['response']) - else: - print("\nFailed to get VLM response. Is 'llava' model pulled and running?") - - except Exception as e: - print(f"An error occurred: {e}") \ No newline at end of file diff --git a/rag_system/utils/validate_model_config.py b/rag_system/utils/validate_model_config.py deleted file mode 100644 index 6516b497..00000000 --- a/rag_system/utils/validate_model_config.py +++ /dev/null @@ -1,219 +0,0 @@ -#!/usr/bin/env python3 -""" -Model Configuration Validation Script -===================================== - -This script validates the consolidated model configuration system to ensure: -1. No configuration conflicts exist -2. All model names are consistent across components -3. Models are accessible and properly configured -4. The configuration validation system works correctly - -Run this after making configuration changes to catch issues early. -""" - -import sys -import os -# Add parent directories to path for imports -sys.path.append(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))) - -from rag_system.main import ( - PIPELINE_CONFIGS, - OLLAMA_CONFIG, - EXTERNAL_MODELS, - validate_model_config -) - -def print_header(title: str): - """Print a formatted header.""" - print(f"\n{'='*60}") - print(f"๐Ÿ” {title}") - print(f"{'='*60}") - -def print_section(title: str): - """Print a formatted section header.""" - print(f"\n{'โ”€'*40}") - print(f"๐Ÿ“‹ {title}") - print(f"{'โ”€'*40}") - -def validate_configuration_consistency(): - """Validate that all configurations are consistent.""" - print_header("CONFIGURATION CONSISTENCY VALIDATION") - - errors = [] - - # 1. Check embedding model consistency - print_section("Embedding Model Consistency") - default_embedding = PIPELINE_CONFIGS["default"]["embedding_model_name"] - external_embedding = EXTERNAL_MODELS["embedding_model"] - fast_embedding = PIPELINE_CONFIGS["fast"]["embedding_model_name"] - - print(f"Default Config: {default_embedding}") - print(f"External Models: {external_embedding}") - print(f"Fast Config: {fast_embedding}") - - if default_embedding != external_embedding: - errors.append(f"โŒ Embedding model mismatch: default={default_embedding}, external={external_embedding}") - elif default_embedding != fast_embedding: - errors.append(f"โŒ Embedding model mismatch: default={default_embedding}, fast={fast_embedding}") - else: - print("โœ… Embedding models are consistent") - - # 2. Check reranker model consistency - print_section("Reranker Model Consistency") - default_reranker = PIPELINE_CONFIGS["default"]["reranker"]["model_name"] - external_reranker = EXTERNAL_MODELS["reranker_model"] - - print(f"Default Config: {default_reranker}") - print(f"External Models: {external_reranker}") - - if default_reranker != external_reranker: - errors.append(f"โŒ Reranker model mismatch: default={default_reranker}, external={external_reranker}") - else: - print("โœ… Reranker models are consistent") - - # 3. Check vision model consistency - print_section("Vision Model Consistency") - default_vision = PIPELINE_CONFIGS["default"]["vision_model_name"] - external_vision = EXTERNAL_MODELS["vision_model"] - - print(f"Default Config: {default_vision}") - print(f"External Models: {external_vision}") - - if default_vision != external_vision: - errors.append(f"โŒ Vision model mismatch: default={default_vision}, external={external_vision}") - else: - print("โœ… Vision models are consistent") - - return errors - -def print_model_usage_map(): - """Print a comprehensive map of which models are used where.""" - print_header("MODEL USAGE MAP") - - print_section("๐Ÿค– Ollama Models (Local Inference)") - for model_type, model_name in OLLAMA_CONFIG.items(): - if model_type != "host": - print(f" {model_type.replace('_', ' ').title()}: {model_name}") - - print_section("๐Ÿ”— External Models (HuggingFace/Direct)") - for model_type, model_name in EXTERNAL_MODELS.items(): - print(f" {model_type.replace('_', ' ').title()}: {model_name}") - - print_section("๐Ÿ“ Model Usage by Component") - usage_map = { - "๐Ÿ”ค Text Embedding": { - "Model": EXTERNAL_MODELS["embedding_model"], - "Used In": ["Retrieval Pipeline", "Semantic Cache", "Dense Retrieval", "Late Chunking"], - "Component": "QwenEmbedder (representations.py)" - }, - "๐Ÿง  Text Generation": { - "Model": OLLAMA_CONFIG["generation_model"], - "Used In": ["Agent Loop", "Answer Synthesis", "Query Decomposition", "Verification"], - "Component": "OllamaClient" - }, - "๐Ÿš€ Enrichment/Routing": { - "Model": OLLAMA_CONFIG["enrichment_model"], - "Used In": ["Query Routing", "Document Overview Analysis"], - "Component": "Agent Loop (_route_via_overviews)" - }, - "๐Ÿ”€ Reranking": { - "Model": EXTERNAL_MODELS["reranker_model"], - "Used In": ["Hybrid Search", "Document Reranking", "AI Reranker"], - "Component": "ColBERT (rerankers-lib) or QwenReranker" - }, - "๐Ÿ‘๏ธ Vision": { - "Model": EXTERNAL_MODELS["vision_model"], - "Used In": ["Multimodal Processing", "Image Embeddings"], - "Component": "Vision Pipeline (when enabled)" - } - } - - for model_name, details in usage_map.items(): - print(f"\n{model_name}") - print(f" Model: {details['Model']}") - print(f" Component: {details['Component']}") - print(f" Used In: {', '.join(details['Used In'])}") - -def test_validation_function(): - """Test the built-in validation function.""" - print_header("VALIDATION FUNCTION TEST") - - try: - result = validate_model_config() - if result: - print("โœ… validate_model_config() passed successfully!") - else: - print("โŒ validate_model_config() returned False") - except Exception as e: - print(f"โŒ validate_model_config() failed with error: {e}") - return False - - return True - -def check_pipeline_configurations(): - """Check all pipeline configurations for completeness.""" - print_header("PIPELINE CONFIGURATION COMPLETENESS") - - required_keys = { - "default": ["storage", "retrieval", "embedding_model_name", "reranker"], - "fast": ["storage", "retrieval", "embedding_model_name"] - } - - errors = [] - - for config_name, required in required_keys.items(): - print_section(f"{config_name.title()} Configuration") - config = PIPELINE_CONFIGS.get(config_name, {}) - - for key in required: - if key in config: - print(f" โœ… {key}: {type(config[key]).__name__}") - else: - error_msg = f"โŒ Missing required key '{key}' in {config_name} config" - errors.append(error_msg) - print(f" {error_msg}") - - return errors - -def main(): - """Run all validation checks.""" - print("๐Ÿš€ Starting Model Configuration Validation") - print(f"Python Path: {sys.path[0]}") - - all_errors = [] - - # Run all validation checks - all_errors.extend(validate_configuration_consistency()) - all_errors.extend(check_pipeline_configurations()) - - # Print model usage map - print_model_usage_map() - - # Test validation function - validation_passed = test_validation_function() - - # Final summary - print_header("VALIDATION SUMMARY") - - if all_errors: - print("โŒ VALIDATION FAILED - Issues Found:") - for error in all_errors: - print(f" {error}") - return 1 - elif not validation_passed: - print("โŒ VALIDATION FAILED - validate_model_config() function failed") - return 1 - else: - print("โœ… ALL VALIDATIONS PASSED!") - print("\n๐ŸŽ‰ Your model configuration is consistent and properly structured!") - print("\n๐Ÿ“‹ Summary:") - print(f" โ€ข Embedding Model: {EXTERNAL_MODELS['embedding_model']}") - print(f" โ€ข Generation Model: {OLLAMA_CONFIG['generation_model']}") - print(f" โ€ข Enrichment Model: {OLLAMA_CONFIG['enrichment_model']}") - print(f" โ€ข Reranker Model: {EXTERNAL_MODELS['reranker_model']}") - print(f" โ€ข Vision Model: {EXTERNAL_MODELS['vision_model']}") - return 0 - -if __name__ == "__main__": - sys.exit(main()) \ No newline at end of file diff --git a/rag_system/utils/watsonx_client.py b/rag_system/utils/watsonx_client.py index 1c926346..e1283d6e 100644 --- a/rag_system/utils/watsonx_client.py +++ b/rag_system/utils/watsonx_client.py @@ -4,6 +4,11 @@ from io import BytesIO from PIL import Image +# The token-usage tracker lives with the Ollama client because that is where the +# counts originate; watsonx imports it only to report its own (absent) counts as +# zeros so the per-query summary stays well-formed on either backend. +from rag_system.utils.ollama_client import record_llm_usage + class WatsonXClient: """ @@ -58,27 +63,6 @@ def _image_to_base64(self, image: Image.Image) -> str: image.save(buffered, format="PNG") return base64.b64encode(buffered.getvalue()).decode('utf-8') - def generate_embedding(self, model: str, text: str) -> List[float]: - """ - Generate embeddings using Watson X embedding models. - Note: This requires using Watson X embedding models through the embeddings API. - """ - try: - from ibm_watsonx_ai.foundation_models import Embeddings - - embedding_model = Embeddings( - model_id=model, - credentials=self.credentials, - project_id=self.project_id - ) - - result = embedding_model.embed_query(text) - return result if isinstance(result, list) else [] - - except Exception as e: - print(f"Error generating embedding: {e}") - return [] - def generate_completion( self, model: str, @@ -105,10 +89,19 @@ def generate_completion( """ try: gen_params = {} - + + # Ollama-style options dict (interface parity): only `temperature` + # maps onto TextGenParameters; the rest (num_ctx, โ€ฆ) are + # Ollama-specific and accepted-and-ignored. + options = kwargs.pop('options', None) + if options and options.get('temperature') is not None: + kwargs.setdefault('temperature', options['temperature']) + if kwargs.get('max_tokens'): gen_params['max_new_tokens'] = kwargs['max_tokens'] - if kwargs.get('temperature'): + # `is not None` so an explicit temperature=0 (the deterministic + # pin) is applied instead of dropped as falsy. + if kwargs.get('temperature') is not None: gen_params['temperature'] = kwargs['temperature'] if kwargs.get('top_p'): gen_params['top_p'] = kwargs['top_p'] @@ -126,9 +119,7 @@ def generate_completion( if images: print("Warning: Image support in Watson X may vary by model") - result = model_inference.generate(prompt=prompt) - else: - result = model_inference.generate(prompt=prompt) + result = model_inference.generate(prompt=prompt) generated_text = "" if isinstance(result, dict): @@ -136,12 +127,21 @@ def generate_completion( else: generated_text = str(result) - return { + # roadmap 4.5: the RAG system's token tracker reads Ollama's + # `prompt_eval_count` / `eval_count`. watsonx's SDK does not surface + # comparable per-call counts through this code path, so report zeros + # rather than omitting the keys โ€” a watsonx run then shows an honest + # "0 tokens counted" instead of silently looking like a cache hit. + payload = { 'response': generated_text, 'model': model, - 'done': True + 'done': True, + 'prompt_eval_count': 0, + 'eval_count': 0, } - + record_llm_usage(payload) + return payload + except Exception as e: print(f"Error generating completion: {e}") return {'response': '', 'error': str(e)} @@ -155,22 +155,28 @@ async def generate_completion_async( images: Optional[List[Image.Image]] = None, enable_thinking: Optional[bool] = None, timeout: int = 60, + options: Optional[Dict[str, Any]] = None, **kwargs ) -> Dict[str, Any]: """ Asynchronous version of generate_completion. - + Note: IBM Watson X SDK may not have native async support, so this is a wrapper around the sync version. + + *options* mirrors OllamaClient's; the sync method translates + `temperature` into watsonx generation params and ignores the rest. """ import asyncio - - loop = asyncio.get_event_loop() + + # Inside a coroutine a loop is always running; get_event_loop() is + # deprecated here and can raise when no loop has been set. + loop = asyncio.get_running_loop() return await loop.run_in_executor( None, lambda: self.generate_completion( model, prompt, format=format, images=images, - enable_thinking=enable_thinking, **kwargs + enable_thinking=enable_thinking, options=options, **kwargs ) ) @@ -181,13 +187,19 @@ def stream_completion( *, images: Optional[List[Image.Image]] = None, enable_thinking: Optional[bool] = None, + stats: Optional[Dict[str, Any]] = None, **kwargs ): """ Generator that yields partial response strings as they arrive. - + Note: Watson X streaming support depends on the SDK version and model. + + *stats* mirrors ``OllamaClient.stream_completion`` (roadmap 4.5). watsonx + exposes no per-stream token counts here, so it is filled with zeros. """ + if stats is not None: + stats.update({'prompt_eval_count': 0, 'eval_count': 0, 'done': True}) try: gen_params = {} if kwargs.get('max_tokens'): @@ -218,8 +230,10 @@ def stream_completion( yield generated_text except Exception as e: + # Yielding "" made a failure look like a valid empty content + # chunk; log and stop the iteration instead. print(f"Error in stream_completion: {e}") - yield "" + return if __name__ == '__main__': diff --git a/requirements-docker.txt b/requirements-docker.txt index a1f57884..cb1f423e 100644 --- a/requirements-docker.txt +++ b/requirements-docker.txt @@ -1,31 +1,20 @@ requests python-dotenv -PyPDF2 -colpali-engine PyMuPDF Pillow transformers==4.51.0 torch==2.4.1 torchvision==0.19.1 -lancedb -rank_bm25 -fuzzywuzzy -python-Levenshtein torchaudio +lancedb sentencepiece accelerate docling cachetools numpy -networkx -matplotlib psutil httpx -scikit-learn pandas -sentence_transformers rerankers -nltk -# Standard library modules (no need to install) -# asyncio, logging, json, os, sys, typing, threading, itertools, math, re -# ocrmac - removed for Docker compatibility (macOS-specific) +# ocrmac - removed for Docker compatibility (macOS-specific); docling's bundled +# EasyOCR backend handles OCR on Linux. diff --git a/requirements.txt b/requirements.txt index f768bb66..6b3fa423 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,32 +1,24 @@ requests python-dotenv -PyPDF2 -colpali-engine -requests -python-dotenv -PyPDF2 -colpali-engine PyMuPDF Pillow transformers==4.51.0 torch==2.4.1 torchvision==0.19.1 -lancedb -rank_bm25 -fuzzywuzzy -python-Levenshtein torchaudio +lancedb sentencepiece accelerate docling cachetools numpy -networkx -matplotlib psutil httpx -scikit-learn pandas -sentence_transformers rerankers -nltk +# Optional: IBM watsonx.ai backend, enabled with LLM_BACKEND=watsonx +# ibm-watsonx-ai>=1.3.39 + +# Optional: Anthropic-API eval judge (eval-only, never in the serving path), +# enabled with JUDGE_MODEL=claude-<model> +# anthropic>=0.121.0 diff --git a/run_system.py b/run_system.py index 8064d6be..117337a8 100644 --- a/run_system.py +++ b/run_system.py @@ -18,6 +18,8 @@ Usage: python run_system.py [--mode dev|prod] [--logs-only] [--no-frontend] + python run_system.py --health + python run_system.py --stop """ import subprocess @@ -36,6 +38,10 @@ from dataclasses import dataclass import psutil +GENERATION_MODEL = os.getenv("GENERATION_MODEL", "qwen3.5:9b") +ENRICHMENT_MODEL = os.getenv("ENRICHMENT_MODEL", "qwen3.5:4b") + + @dataclass class ServiceConfig: name: str @@ -46,6 +52,7 @@ class ServiceConfig: health_check_path: str = "/health" startup_delay: int = 2 required: bool = True + build_command: Optional[List[str]] = None class ColoredFormatter(logging.Formatter): """Custom formatter with colors for different log levels and services.""" @@ -89,11 +96,12 @@ def __init__(self, mode: str = "dev", logs_dir: str = "logs"): self.mode = mode self.logs_dir = Path(logs_dir) self.logs_dir.mkdir(exist_ok=True) - + self.pidfile = self.logs_dir / 'run_system.pid' + self.processes: Dict[str, subprocess.Popen] = {} self.log_threads: Dict[str, threading.Thread] = {} self.running = False - + # Setup logging self.setup_logging() @@ -129,6 +137,7 @@ def _get_service_configs(self) -> Dict[str, ServiceConfig]: name='ollama', command=['ollama', 'serve'], port=11434, + health_check_path='/api/tags', startup_delay=5, required=True ), @@ -150,19 +159,21 @@ def _get_service_configs(self) -> Dict[str, ServiceConfig]: name='frontend', command=['npm', 'run', 'dev' if self.mode == 'dev' else 'start'], port=3000, + health_check_path='/', startup_delay=5, required=False # Optional in case Node.js not available ) } - + # Production mode adjustments if self.mode == 'prod': - # Use production build for frontend + # `next start` needs an existing .next build, so build before starting base_configs['frontend'].command = ['npm', 'run', 'start'] + base_configs['frontend'].build_command = ['npm', 'run', 'build'] # Add production environment variables base_configs['rag-api'].env = {'NODE_ENV': 'production'} base_configs['backend'].env = {'NODE_ENV': 'production'} - + return base_configs def _signal_handler(self, signum, frame): @@ -223,8 +234,8 @@ def ensure_models(self): """Ensure required Ollama models are available.""" self.logger.info("๐Ÿ“ฅ Checking required models...") - required_models = ['qwen3:8b', 'qwen3:0.6b'] - + required_models = [GENERATION_MODEL, ENRICHMENT_MODEL] + try: # Get list of installed models result = subprocess.run(['ollama', 'list'], @@ -256,15 +267,20 @@ def start_service(self, service_name: str, config: ServiceConfig) -> bool: self.logger.warning(f"โš ๏ธ Port {config.port} already in use, skipping {service_name}") return not config.required + if config.build_command and not self._run_build(service_name, config): + return False + self.logger.info(f"๐Ÿ”„ Starting {service_name} on port {config.port}...") - + try: # Setup environment env = os.environ.copy() if config.env: env.update(config.env) - - # Start process + + # Start process in its own session/process group (POSIX; ignored on + # Windows) so _stop_service can signal the whole tree โ€” wrappers + # like npm spawn the real server as a child. process = subprocess.Popen( config.command, cwd=config.cwd, @@ -273,7 +289,8 @@ def start_service(self, service_name: str, config: ServiceConfig) -> bool: stderr=subprocess.STDOUT, text=True, bufsize=1, - universal_newlines=True + universal_newlines=True, + start_new_session=True ) self.processes[service_name] = process @@ -293,15 +310,56 @@ def start_service(self, service_name: str, config: ServiceConfig) -> bool: # Check if process is still running if process.poll() is None: self.logger.info(f"โœ… {service_name} started successfully (PID: {process.pid})") + self._write_pidfile() return True else: self.logger.error(f"โŒ {service_name} failed to start") + del self.processes[service_name] return False - + except Exception as e: self.logger.error(f"โŒ Failed to start {service_name}: {e}") + self.processes.pop(service_name, None) return False - + + def _run_build(self, service_name: str, config: ServiceConfig) -> bool: + """Run a service's build step before starting it (prod frontend needs `next build`).""" + self.logger.info(f"๐Ÿ—๏ธ Building {service_name}: {' '.join(config.build_command)}") + try: + result = subprocess.run(config.build_command, cwd=config.cwd) + except FileNotFoundError as e: + self.logger.error(f"โŒ Build command for {service_name} not found: {e}") + return False + if result.returncode != 0: + self.logger.error(f"โŒ Build failed for {service_name} (exit {result.returncode})") + return False + self.logger.info(f"โœ… {service_name} build complete") + return True + + def _write_pidfile(self): + """Persist launcher + child PIDs so `--stop` can find them in another shell.""" + data = { + 'launcher': os.getpid(), + 'services': {name: proc.pid for name, proc in self.processes.items()}, + } + try: + self.pidfile.write_text(json.dumps(data, indent=2)) + except OSError as e: + self.logger.warning(f"โš ๏ธ Could not write pidfile {self.pidfile}: {e}") + + def _read_pidfile(self) -> Dict: + try: + return json.loads(self.pidfile.read_text()) + except (OSError, ValueError): + return {} + + def _clear_pidfile(self): + try: + self.pidfile.unlink() + except OSError: + pass + + def _monitor_service_logs(self, service_name: str, process: subprocess.Popen): """Monitor service logs and forward to main logger.""" service_logger = logging.getLogger(service_name) @@ -335,14 +393,43 @@ def _monitor_service_logs(self, service_name: str, process: subprocess.Popen): self.logger.error(f"Error monitoring {service_name} logs: {e}") def health_check(self, service_name: str, config: ServiceConfig) -> bool: - """Perform health check on a service.""" + """Perform an HTTP health check against a service.""" + url = f"http://localhost:{config.port}{config.health_check_path}" try: - url = f"http://localhost:{config.port}{config.health_check_path}" response = requests.get(url, timeout=5) return response.status_code == 200 - except: + except requests.exceptions.RequestException: return False - + + def print_health_report(self) -> bool: + """Run a real health check per service and report pass/fail.""" + self.logger.info("๐Ÿฅ Health check:") + all_required_healthy = True + + for service_name, config in self.services.items(): + url = f"http://localhost:{config.port}{config.health_check_path}" + if not self.is_port_in_use(config.port): + status = "โŒ Not running" + healthy = False + elif self.health_check(service_name, config): + status = "โœ… Healthy" + healthy = True + else: + status = "โš ๏ธ Port open, health check failed" + healthy = False + + self.logger.info(f" โ€ข {service_name.capitalize():<10}: {status:<32} {url}") + if config.required and not healthy: + all_required_healthy = False + + self.logger.info("") + if all_required_healthy: + self.logger.info("โœ… All required services are healthy") + else: + self.logger.error("โŒ One or more required services are unhealthy") + return all_required_healthy + + def start_all(self, skip_frontend: bool = False) -> bool: """Start all services in order.""" self.logger.info("๐Ÿš€ Starting RAG System Components...") @@ -421,46 +508,149 @@ def _print_status_summary(self): self.logger.info("๐ŸŒ Access your RAG system at: http://localhost:3000") self.logger.info("") self.logger.info("๐Ÿ“‹ Useful commands:") - self.logger.info(" โ€ข Stop system: Ctrl+C") + self.logger.info(" โ€ข Stop system: Ctrl+C (or: python run_system.py --stop)") self.logger.info(" โ€ข Check logs: tail -f logs/*.log") self.logger.info(" โ€ข Health check: python run_system.py --health") - + + def stop_all(self) -> bool: + """Stop services recorded in the pidfile (works from a separate shell).""" + record = self._read_pidfile() + if not record: + self.logger.warning(f"โš ๏ธ No pidfile at {self.pidfile} โ€“ nothing to stop.") + return False + + stopped = 0 + + # Stop the launcher first so its monitor loop cannot restart anything + launcher_pid = record.get('launcher') + if launcher_pid and launcher_pid != os.getpid(): + self._terminate_pid('launcher', launcher_pid) + + for service_name, pid in (record.get('services') or {}).items(): + if self._terminate_pid(service_name, pid): + stopped += 1 + + self._clear_pidfile() + self.logger.info(f"โœ… Stopped {stopped} service(s)") + return stopped > 0 + + def _terminate_pid(self, label: str, pid: int) -> bool: + """Terminate a PID, escalating to kill, and reap its children.""" + try: + process = psutil.Process(pid) + except psutil.NoSuchProcess: + self.logger.info(f" โ€ข {label}: already stopped (pid {pid})") + return False + except psutil.Error as e: + self.logger.warning(f"โš ๏ธ Could not inspect {label} (pid {pid}): {e}") + return False + + targets = [process] + try: + targets.extend(process.children(recursive=True)) + except psutil.Error: + pass + + for target in reversed(targets): + try: + target.terminate() + except psutil.Error: + pass + + _, alive = psutil.wait_procs(targets, timeout=10) + for target in alive: + try: + target.kill() + except psutil.Error: + pass + + self.logger.info(f" โ€ข {label}: stopped (pid {pid})") + return True + + def tail_logs(self): + """Follow every logs/*.log file, like `tail -f logs/*.log`.""" + log_files = sorted(self.logs_dir.glob('*.log')) + if not log_files: + self.logger.warning(f"โš ๏ธ No log files in {self.logs_dir}/ โ€“ start the system first.") + return + + self.logger.info(f"๐Ÿ“‹ Following {len(log_files)} log file(s): {', '.join(f.name for f in log_files)}") + handles = {} + try: + for path in log_files: + handle = path.open('r', errors='replace') + handle.seek(0, os.SEEK_END) + handles[path.stem] = handle + + while True: + emitted = False + for name, handle in handles.items(): + line = handle.readline() + while line: + print(f"[{name.upper()}] {line.rstrip()}") + emitted = True + line = handle.readline() + if not emitted: + time.sleep(0.5) + except KeyboardInterrupt: + self.logger.info("Log tailing stopped by user") + finally: + for handle in handles.values(): + handle.close() + def shutdown(self): """Gracefully shutdown all services.""" if not self.running: return - + self.logger.info("๐Ÿ›‘ Shutting down RAG system...") self.running = False - + # Stop services in reverse order for service_name in reversed(list(self.processes.keys())): self._stop_service(service_name) - + + self._clear_pidfile() self.logger.info("โœ… All services stopped") - + + + def _signal_process_tree(self, process: subprocess.Popen, sig: int): + """Signal the service's whole process group (POSIX), else just the child.""" + if os.name == 'posix': + try: + os.killpg(process.pid, sig) + except ProcessLookupError: + pass # process group already gone + elif sig == signal.SIGTERM: + process.terminate() + else: + process.kill() + def _stop_service(self, service_name: str): """Stop a single service.""" if service_name not in self.processes: return - + process = self.processes[service_name] self.logger.info(f"๐Ÿ”„ Stopping {service_name}...") - + try: - # Try graceful shutdown first - process.terminate() - + # Try graceful shutdown first. Services are spawned with + # start_new_session=True, so signal the process GROUP: terminate() + # alone would kill only the direct child (e.g. npm) and orphan the + # real server it spawned. + self._signal_process_tree(process, signal.SIGTERM) + # Wait up to 10 seconds for graceful shutdown try: process.wait(timeout=10) except subprocess.TimeoutExpired: # Force kill if graceful shutdown fails - process.kill() + self._signal_process_tree(process, signal.SIGKILL) process.wait() - + self.logger.info(f"โœ… {service_name} stopped") - + except Exception as e: self.logger.error(f"โŒ Error stopping {service_name}: {e}") finally: @@ -509,22 +699,20 @@ def main(): try: if args.health: - # Health check mode - manager._print_status_summary() - return - + # Health check mode - real HTTP checks against each /health endpoint + sys.exit(0 if manager.print_health_report() else 1) + if args.stop: - # Stop mode - kill any running processes + # Stop mode - terminate the processes recorded in the pidfile manager.logger.info("๐Ÿ›‘ Stopping all RAG system processes...") - # Implementation for stopping would go here - return - + sys.exit(0 if manager.stop_all() else 1) + if args.logs_only: - # Logs only mode - just tail existing logs + # Logs only mode - tail the log files written by a running launcher manager.logger.info("๐Ÿ“‹ Showing aggregated logs... (Press Ctrl+C to stop)") - manager.monitor() + manager.tail_logs() return - + # Normal startup mode if manager.start_all(skip_frontend=args.no_frontend): manager.monitor() diff --git a/setup_rag_system.sh b/setup_rag_system.sh deleted file mode 100644 index aab81518..00000000 --- a/setup_rag_system.sh +++ /dev/null @@ -1,520 +0,0 @@ -#!/bin/bash -# setup_rag_system.sh - Complete RAG System Setup Script -# This script handles Docker installation, system setup, and initial configuration - -set -e - -# Colors for output -RED='\033[0;31m' -GREEN='\033[0;32m' -YELLOW='\033[1;33m' -BLUE='\033[0;34m' -NC='\033[0m' # No Color - -# Logging function -log() { - echo -e "${GREEN}[$(date +'%Y-%m-%d %H:%M:%S')] $1${NC}" -} - -warn() { - echo -e "${YELLOW}[$(date +'%Y-%m-%d %H:%M:%S')] WARNING: $1${NC}" -} - -error() { - echo -e "${RED}[$(date +'%Y-%m-%d %H:%M:%S')] ERROR: $1${NC}" -} - -info() { - echo -e "${BLUE}[$(date +'%Y-%m-%d %H:%M:%S')] INFO: $1${NC}" -} - -# Check if running as root -if [[ $EUID -eq 0 ]]; then - error "This script should not be run as root (except for package installation steps)" - exit 1 -fi - -echo "================================================================" -echo "๐Ÿš€ RAG System Complete Setup Script" -echo "================================================================" -echo "" - -# Step 1: System Requirements Check -log "Step 1: Checking system requirements..." - -# Check OS -if [[ "$OSTYPE" == "darwin"* ]]; then - OS="macos" - info "Detected macOS" -elif [[ -f /etc/os-release ]]; then - . /etc/os-release - OS=$ID - info "Detected Linux: $OS" -else - error "Unsupported operating system" - exit 1 -fi - -# Check available memory -MEMORY_GB=$(free -g 2>/dev/null | grep '^Mem:' | awk '{print $2}' || sysctl -n hw.memsize 2>/dev/null | awk '{print int($1/1024/1024/1024)}' || echo "unknown") -if [[ "$MEMORY_GB" != "unknown" && "$MEMORY_GB" -lt 8 ]]; then - warn "System has ${MEMORY_GB}GB RAM. Recommended: 16GB+ for optimal performance" -else - info "Memory check passed: ${MEMORY_GB}GB RAM" -fi - -# Check available disk space -DISK_GB=$(df -BG . | tail -1 | awk '{print $4}' | sed 's/G//' || echo "unknown") -if [[ "$DISK_GB" != "unknown" && "$DISK_GB" -lt 50 ]]; then - warn "Available disk space: ${DISK_GB}GB. Recommended: 50GB+ free space" -else - info "Disk space check passed: ${DISK_GB}GB available" -fi - -# Step 2: Install Dependencies -log "Step 2: Installing system dependencies..." - -# Install Git if not present -if ! command -v git &> /dev/null; then - info "Installing Git..." - case $OS in - "macos") - if command -v brew &> /dev/null; then - brew install git - else - error "Git not found. Please install Git first or install Homebrew" - exit 1 - fi - ;; - "ubuntu"|"debian") - sudo apt-get update - sudo apt-get install -y git - ;; - "centos"|"rhel"|"fedora") - if command -v dnf &> /dev/null; then - sudo dnf install -y git - else - sudo yum install -y git - fi - ;; - esac -else - info "Git is already installed: $(git --version)" -fi - -# Install curl if not present -if ! command -v curl &> /dev/null; then - info "Installing curl..." - case $OS in - "macos") - # curl is usually pre-installed on macOS - ;; - "ubuntu"|"debian") - sudo apt-get install -y curl - ;; - "centos"|"rhel"|"fedora") - if command -v dnf &> /dev/null; then - sudo dnf install -y curl - else - sudo yum install -y curl - fi - ;; - esac -else - info "curl is already installed" -fi - -# Step 3: Install Docker -log "Step 3: Installing Docker..." - -if command -v docker &> /dev/null; then - info "Docker is already installed: $(docker --version)" -else - info "Docker not found. Installing Docker..." - - case $OS in - "macos") - # Check if Homebrew is installed - if ! command -v brew &> /dev/null; then - info "Installing Homebrew..." - /bin/bash -c "$(curl -fsSL https://raw.githubusercontent.com/Homebrew/install/HEAD/install.sh)" - fi - - # Install Docker Desktop - info "Installing Docker Desktop..." - brew install --cask docker - - warn "Docker Desktop installed. Please:" - warn "1. Start Docker Desktop from Applications" - warn "2. Wait for Docker to start completely" - warn "3. Run this script again" - exit 0 - ;; - - "ubuntu"|"debian") - # Update package index - sudo apt-get update - - # Install dependencies - sudo apt-get install -y \ - ca-certificates \ - curl \ - gnupg \ - lsb-release - - # Add Docker's official GPG key - sudo mkdir -p /etc/apt/keyrings - curl -fsSL https://download.docker.com/linux/$OS/gpg | sudo gpg --dearmor -o /etc/apt/keyrings/docker.gpg - - # Set up repository - echo \ - "deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.gpg] https://download.docker.com/linux/$OS \ - $(lsb_release -cs) stable" | sudo tee /etc/apt/sources.list.d/docker.list > /dev/null - - # Install Docker Engine - sudo apt-get update - sudo apt-get install -y docker-ce docker-ce-cli containerd.io docker-compose-plugin - - # Add user to docker group - sudo usermod -aG docker $USER - - # Start Docker service - sudo systemctl enable docker - sudo systemctl start docker - - info "Docker installed successfully!" - warn "Please log out and log back in for group changes to take effect, then run this script again" - warn "Or run: newgrp docker && $0" - exit 0 - ;; - - "centos"|"rhel"|"fedora") - # Install required packages - if command -v dnf &> /dev/null; then - sudo dnf install -y yum-utils - sudo dnf config-manager --add-repo https://download.docker.com/linux/centos/docker-ce.repo - sudo dnf install -y docker-ce docker-ce-cli containerd.io docker-compose-plugin - else - sudo yum install -y yum-utils - sudo yum-config-manager --add-repo https://download.docker.com/linux/centos/docker-ce.repo - sudo yum install -y docker-ce docker-ce-cli containerd.io docker-compose-plugin - fi - - # Add user to docker group - sudo usermod -aG docker $USER - - # Start Docker service - sudo systemctl enable docker - sudo systemctl start docker - - info "Docker installed successfully!" - warn "Please log out and log back in for group changes to take effect, then run this script again" - exit 0 - ;; - esac -fi - -# Verify Docker is working -if ! docker --version &> /dev/null; then - error "Docker is not working properly. Please check Docker installation" - exit 1 -fi - -if ! docker compose version &> /dev/null; then - error "Docker Compose is not working properly. Please check Docker Compose installation" - exit 1 -fi - -info "Docker verification passed: $(docker --version)" -info "Docker Compose verification passed: $(docker compose version)" - -# Test Docker daemon -if ! docker ps &> /dev/null; then - error "Cannot connect to Docker daemon. Please ensure Docker is running" - exit 1 -fi - -# Step 4: Setup RAG System -log "Step 4: Setting up RAG System..." - -# Create project directory structure -info "Creating directory structure..." -mkdir -p {lancedb,shared_uploads,logs,ollama_data} -mkdir -p index_store/{overviews,bm25,graph} -mkdir -p backups - -# Set proper permissions -chmod 755 {lancedb,shared_uploads,logs,ollama_data} -chmod 755 index_store/{overviews,bm25,graph} -chmod 755 backups - -# Create environment file -if [[ ! -f ".env" ]]; then - info "Creating environment configuration..." - cat > .env << 'EOF' -# System Configuration -NODE_ENV=production -LOG_LEVEL=info -DEBUG=false - -# Service URLs -FRONTEND_URL=http://localhost:3000 -BACKEND_URL=http://localhost:8000 -RAG_API_URL=http://localhost:8001 -OLLAMA_URL=http://localhost:11434 - -# Database Configuration -DATABASE_PATH=./backend/chat_data.db -LANCEDB_PATH=./lancedb -UPLOADS_PATH=./shared_uploads -INDEX_STORE_PATH=./index_store - -# Model Configuration -DEFAULT_EMBEDDING_MODEL=sentence-transformers/all-mpnet-base-v2 -# Default model names - updated to current versions -DEFAULT_GENERATION_MODEL=qwen3:8b -DEFAULT_RERANKER_MODEL=answerdotai/answerai-colbert-small-v1 -DEFAULT_ENRICHMENT_MODEL=qwen3:0.6b - -# Performance Configuration -MAX_CONCURRENT_REQUESTS=5 -REQUEST_TIMEOUT=300 -EMBEDDING_BATCH_SIZE=32 -MAX_CONTEXT_LENGTH=4096 - -# Security Configuration -CORS_ORIGINS=http://localhost:3000 -API_KEY_REQUIRED=false -RATE_LIMIT_REQUESTS=100 -RATE_LIMIT_WINDOW=60 - -# Storage Configuration -MAX_FILE_SIZE=50MB -MAX_UPLOAD_FILES=10 -CLEANUP_INTERVAL=3600 -BACKUP_RETENTION_DAYS=30 -EOF - info "Environment file created: .env" -else - info "Environment file already exists: .env" -fi - -# Step 5: Build and Start Services -log "Step 5: Building and starting services..." - -info "Building Docker containers (this may take 10-15 minutes)..." -docker compose build --no-cache - -info "Starting services..." -docker compose up -d - -# Wait for services to start -info "Waiting for services to initialize..." -sleep 30 - -# Check service status -info "Checking service status..." -docker compose ps - -# Step 6: Install AI Models -log "Step 6: Installing AI models..." - -# Wait for Ollama to be ready -info "Waiting for Ollama to be ready..." -max_attempts=30 -attempt=0 -while ! docker compose exec ollama ollama list &> /dev/null; do - if [ $attempt -ge $max_attempts ]; then - error "Ollama failed to start after $max_attempts attempts" - exit 1 - fi - info "Waiting for Ollama... (attempt $((attempt+1))/$max_attempts)" - sleep 10 - ((attempt++)) -done - -# Download Ollama models -info "Downloading required Ollama models..." -docker compose exec ollama ollama pull qwen3:8b -docker compose exec ollama ollama pull qwen3:0.6b - -info "Verifying model installation..." -docker compose exec ollama ollama list - -# Step 7: System Verification -log "Step 7: Verifying system installation..." - -# Check service health -info "Checking service health..." -services=("frontend:3000" "backend:8000" "rag-api:8001" "ollama:11434") -for service in "${services[@]}"; do - name="${service%:*}" - port="${service#*:}" - - if curl -s -f "http://localhost:$port" &> /dev/null || curl -s -f "http://localhost:$port/health" &> /dev/null || curl -s -f "http://localhost:$port/api/tags" &> /dev/null || curl -s -f "http://localhost:$port/models" &> /dev/null; then - info "โœ… $name service is healthy" - else - warn "โš ๏ธ $name service may not be ready yet" - fi -done - -# Step 8: Create Helper Scripts -log "Step 8: Creating helper scripts..." - -# Create start script -cat > start_rag_system.sh << 'EOF' -#!/bin/bash -# Start RAG System -echo "Starting RAG System..." -docker compose up -d -echo "RAG System started. Access at: http://localhost:3000" -EOF -chmod +x start_rag_system.sh - -# Create stop script -cat > stop_rag_system.sh << 'EOF' -#!/bin/bash -# Stop RAG System -echo "Stopping RAG System..." -docker compose down -echo "RAG System stopped." -EOF -chmod +x stop_rag_system.sh - -# Create status script -cat > status_rag_system.sh << 'EOF' -#!/bin/bash -# Check RAG System Status -echo "=== RAG System Status ===" -docker compose ps -echo "" -echo "=== Service Health ===" -curl -s -f http://localhost:3000 && echo "โœ… Frontend: OK" || echo "โŒ Frontend: FAIL" -curl -s -f http://localhost:8000/health && echo "โœ… Backend: OK" || echo "โŒ Backend: FAIL" -curl -s -f http://localhost:8001/models && echo "โœ… RAG API: OK" || echo "โŒ RAG API: FAIL" -curl -s -f http://localhost:11434/api/tags && echo "โœ… Ollama: OK" || echo "โŒ Ollama: FAIL" -EOF -chmod +x status_rag_system.sh - -# Create backup script -cat > backup_rag_system.sh << 'EOF' -#!/bin/bash -# Backup RAG System Data -BACKUP_DIR="./backups/$(date +%Y%m%d_%H%M%S)" -mkdir -p "$BACKUP_DIR" - -echo "Creating backup in $BACKUP_DIR..." - -# Stop services -docker compose down - -# Backup data -cp -r ./backend/chat_data.db "$BACKUP_DIR/" 2>/dev/null || true -cp -r ./lancedb "$BACKUP_DIR/" 2>/dev/null || true -cp -r ./shared_uploads "$BACKUP_DIR/" 2>/dev/null || true -cp -r ./index_store "$BACKUP_DIR/" 2>/dev/null || true - -# Backup configuration -cp .env "$BACKUP_DIR/" -cp docker-compose.yml "$BACKUP_DIR/" - -# Restart services -docker compose up -d - -echo "Backup completed: $BACKUP_DIR" -EOF -chmod +x backup_rag_system.sh - -# Create update script -cat > update_rag_system.sh << 'EOF' -#!/bin/bash -# Update RAG System -echo "Updating RAG System..." - -# Backup first -./backup_rag_system.sh - -# Pull latest changes -git pull origin main - -# Rebuild containers -docker compose build --no-cache - -# Restart services -docker compose up -d - -echo "Update completed!" -EOF -chmod +x update_rag_system.sh - -info "Helper scripts created:" -info " - start_rag_system.sh: Start the system" -info " - stop_rag_system.sh: Stop the system" -info " - status_rag_system.sh: Check system status" -info " - backup_rag_system.sh: Backup system data" -info " - update_rag_system.sh: Update the system" - -# Step 9: Final Setup -log "Step 9: Final setup and verification..." - -# Create initial database if it doesn't exist -if [[ ! -f "./backend/chat_data.db" ]]; then - info "Creating initial database..." - docker compose exec backend python -c " -import sqlite3 -conn = sqlite3.connect('/app/backend/chat_data.db') -conn.execute('CREATE TABLE IF NOT EXISTS sessions (id TEXT PRIMARY KEY, title TEXT, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP)') -conn.execute('CREATE TABLE IF NOT EXISTS messages (id INTEGER PRIMARY KEY, session_id TEXT, content TEXT, role TEXT, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP)') -conn.execute('CREATE TABLE IF NOT EXISTS indexes (id TEXT PRIMARY KEY, name TEXT, metadata TEXT, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP)') -conn.execute('CREATE TABLE IF NOT EXISTS session_indexes (session_id TEXT, index_id TEXT, PRIMARY KEY (session_id, index_id))') -conn.commit() -conn.close() -print('Database initialized') -" 2>/dev/null || warn "Database initialization may have failed" -fi - -# Final health check -info "Performing final health check..." -sleep 10 -./status_rag_system.sh - -echo "" -echo "================================================================" -echo "๐ŸŽ‰ RAG System Setup Complete!" -echo "================================================================" -echo "" -echo "โœ… System Status:" -echo " - Frontend: http://localhost:3000" -echo " - Backend API: http://localhost:8000" -echo " - RAG API: http://localhost:8001" -echo " - Ollama: http://localhost:11434" -echo "" -echo "๐Ÿ“š Documentation:" -echo " - System Overview: Documentation/system_overview.md" -echo " - Deployment Guide: Documentation/deployment_guide.md" -echo " - Docker Usage: Documentation/docker_usage.md" -echo " - Installation Guide: Documentation/installation_guide.md" -echo "" -echo "๐Ÿ”ง Helper Scripts:" -echo " - Start system: ./start_rag_system.sh" -echo " - Stop system: ./stop_rag_system.sh" -echo " - Check status: ./status_rag_system.sh" -echo " - Backup data: ./backup_rag_system.sh" -echo " - Update system: ./update_rag_system.sh" -echo "" -echo "๐Ÿš€ Next Steps:" -echo " 1. Open http://localhost:3000 in your browser" -echo " 2. Create a new chat session" -echo " 3. Upload some PDF documents" -echo " 4. Start asking questions about your documents!" -echo "" -echo "๐Ÿ“‹ System Information:" -echo " - OS: $OS" -echo " - Memory: ${MEMORY_GB}GB" -echo " - Disk Space: ${DISK_GB}GB available" -echo " - Docker: $(docker --version)" -echo " - Docker Compose: $(docker compose version)" -echo "" -echo "For support and troubleshooting, check the documentation in the" -echo "Documentation/ folder or run ./status_rag_system.sh to check system health." -echo "" \ No newline at end of file diff --git a/simple_create_index.sh b/simple_create_index.sh deleted file mode 100755 index ebe5b845..00000000 --- a/simple_create_index.sh +++ /dev/null @@ -1,228 +0,0 @@ -#!/bin/bash - -# Simple Index Creation Script for LocalGPT RAG System -# Usage: ./simple_create_index.sh "Index Name" "path/to/document.pdf" [additional_files...] - -set -e # Exit on any error - -# Colors for output -RED='\033[0;31m' -GREEN='\033[0;32m' -YELLOW='\033[1;33m' -BLUE='\033[0;34m' -NC='\033[0m' # No Color - -# Function to print colored output -print_status() { - echo -e "${BLUE}[INFO]${NC} $1" -} - -print_success() { - echo -e "${GREEN}[SUCCESS]${NC} $1" -} - -print_warning() { - echo -e "${YELLOW}[WARNING]${NC} $1" -} - -print_error() { - echo -e "${RED}[ERROR]${NC} $1" -} - -# Function to check if a command exists -command_exists() { - command -v "$1" >/dev/null 2>&1 -} - -# Function to check prerequisites -check_prerequisites() { - print_status "Checking prerequisites..." - - # Check Python - if ! command_exists python3; then - print_error "Python 3 is required but not installed." - exit 1 - fi - - # Check if we're in the right directory - if [ ! -f "run_system.py" ] || [ ! -d "rag_system" ]; then - print_error "This script must be run from the LocalGPT project root directory." - exit 1 - fi - - # Check if Ollama is running - if ! curl -s http://localhost:11434/api/tags >/dev/null 2>&1; then - print_error "Ollama is not running. Please start Ollama first:" - echo " ollama serve" - exit 1 - fi - - print_success "Prerequisites check passed" -} - -# Function to validate documents -validate_documents() { - local documents=("$@") - local valid_docs=() - - print_status "Validating documents..." - - for doc in "${documents[@]}"; do - if [ -f "$doc" ]; then - # Check file extension - case "${doc##*.}" in - pdf|txt|docx|md|html|htm) - valid_docs+=("$doc") - print_status "โœ“ Valid document: $doc" - ;; - *) - print_warning "Unsupported file type: $doc (skipping)" - ;; - esac - else - print_warning "File not found: $doc (skipping)" - fi - done - - if [ ${#valid_docs[@]} -eq 0 ]; then - print_error "No valid documents found." - exit 1 - fi - - echo "${valid_docs[@]}" -} - -# Function to create index using Python -create_index() { - local index_name="$1" - shift - local documents=("$@") - - print_status "Creating index: $index_name" - print_status "Documents: ${documents[*]}" - - # Create a temporary Python script to create the index - cat > /tmp/create_index_temp.py << EOF -#!/usr/bin/env python3 -import sys -import os -import json -sys.path.insert(0, os.getcwd()) - -from rag_system.main import PIPELINE_CONFIGS -from rag_system.pipelines.indexing_pipeline import IndexingPipeline -from rag_system.utils.ollama_client import OllamaClient -from backend.database import ChatDatabase -import uuid - -def create_index_simple(): - try: - # Initialize database - db = ChatDatabase() - - # Create index record - index_id = db.create_index( - name="$index_name", - description="Created with simple_create_index.sh", - metadata={ - "chunk_size": 512, - "chunk_overlap": 64, - "enable_enrich": True, - "enable_latechunk": True, - "retrieval_mode": "hybrid", - "created_by": "simple_create_index.sh" - } - ) - - # Add documents to index - documents = [${documents[@]/#/\"} ${documents[@]/%/\"}] - for doc_path in documents: - if doc_path.strip(): # Skip empty strings - filename = os.path.basename(doc_path.strip()) - db.add_document_to_index(index_id, filename, os.path.abspath(doc_path.strip())) - - # Initialize pipeline - config = PIPELINE_CONFIGS.get("default", {}) - ollama_client = OllamaClient() - ollama_config = { - "generation_model": "qwen3:0.6b", - "embedding_model": "qwen3:0.6b" - } - - pipeline = IndexingPipeline(config, ollama_client, ollama_config) - - # Process documents - valid_docs = [doc.strip() for doc in documents if doc.strip() and os.path.exists(doc.strip())] - if valid_docs: - pipeline.process_documents(valid_docs) - - print(f"โœ… Index '{index_name}' created successfully!") - print(f"Index ID: {index_id}") - print(f"Processed {len(valid_docs)} documents") - - return index_id - - except Exception as e: - print(f"โŒ Error creating index: {e}") - import traceback - traceback.print_exc() - return None - -if __name__ == "__main__": - create_index_simple() -EOF - - # Run the Python script - python3 /tmp/create_index_temp.py - - # Clean up - rm -f /tmp/create_index_temp.py -} - -# Function to show usage -show_usage() { - echo "Usage: $0 \"Index Name\" \"path/to/document.pdf\" [additional_files...]" - echo "" - echo "Examples:" - echo " $0 \"My Documents\" \"document.pdf\"" - echo " $0 \"Research Papers\" \"paper1.pdf\" \"paper2.pdf\" \"notes.txt\"" - echo " $0 \"Invoice Collection\" ./invoices/*.pdf" - echo "" - echo "Supported file types: PDF, TXT, DOCX, MD, HTML" -} - -# Main script -main() { - # Check arguments - if [ $# -lt 2 ]; then - print_error "Insufficient arguments provided." - show_usage - exit 1 - fi - - local index_name="$1" - shift - local documents=("$@") - - # Check prerequisites - check_prerequisites - - # Validate documents - local valid_documents - valid_documents=($(validate_documents "${documents[@]}")) - - if [ ${#valid_documents[@]} -eq 0 ]; then - print_error "No valid documents to process." - exit 1 - fi - - # Create the index - print_status "Starting index creation process..." - create_index "$index_name" "${valid_documents[@]}" - - print_success "Index creation completed!" - print_status "You can now use the index in the LocalGPT interface." -} - -# Run main function with all arguments -main "$@" \ No newline at end of file diff --git a/src/app/layout.tsx b/src/app/layout.tsx index 2de7fc0a..083f2c00 100644 --- a/src/app/layout.tsx +++ b/src/app/layout.tsx @@ -13,8 +13,8 @@ const geistMono = Geist_Mono({ }); export const metadata: Metadata = { - title: "Create Next App", - description: "Generated by create next app", + title: "localGPT", + description: "Chat with your local documents โ€” a fully local RAG assistant.", }; export default function RootLayout({ diff --git a/src/components/IndexForm.tsx b/src/components/IndexForm.tsx index f0e0d9af..0f88faa1 100644 --- a/src/components/IndexForm.tsx +++ b/src/components/IndexForm.tsx @@ -4,7 +4,7 @@ import { GlassInput } from '@/components/ui/GlassInput'; import { GlassToggle } from '@/components/ui/GlassToggle'; import { AccordionGroup } from '@/components/ui/AccordionGroup'; import { ModelSelect } from '@/components/ModelSelect'; -import { chatAPI, ChatSession } from '@/lib/api'; +import { chatAPI, ChatSession, DEFAULT_ENRICHMENT_MODEL } from '@/lib/api'; import { InfoTooltip } from '@/components/ui/InfoTooltip'; interface Props { @@ -16,16 +16,14 @@ export function IndexForm({ onClose, onIndexed }: Props) { const [files, setFiles] = useState<FileList | null>(null); const [indexName, setIndexName] = useState(''); const [chunkSize, setChunkSize] = useState(512); - const [chunkOverlap, setChunkOverlap] = useState(64); const [windowSize, setWindowSize] = useState(5); const [enableEnrich, setEnableEnrich] = useState(true); - const [retrievalMode, setRetrievalMode] = useState<'hybrid' | 'vector' | 'fts'>('hybrid'); + const [retrievalMode, setRetrievalMode] = useState<'hybrid' | 'vector_only' | 'fts_only'>('hybrid'); const [embeddingModel, setEmbeddingModel] = useState<string>(); - const DEFAULT_LLM = 'qwen3:0.6b'; - const [enrichModel, setEnrichModel] = useState<string>(DEFAULT_LLM); - const [overviewModel, setOverviewModel] = useState<string>(DEFAULT_LLM); - const [batchSizeEmbed, setBatchSizeEmbed] = useState(64); - const [batchSizeEnrich, setBatchSizeEnrich] = useState(64); + const [enrichModel, setEnrichModel] = useState<string>(DEFAULT_ENRICHMENT_MODEL); + const [overviewModel, setOverviewModel] = useState<string>(DEFAULT_ENRICHMENT_MODEL); + const [batchSizeEmbed, setBatchSizeEmbed] = useState(50); + const [batchSizeEnrich, setBatchSizeEnrich] = useState(25); const [loading, setLoading] = useState(false); const [enableLateChunk, setEnableLateChunk] = useState(false); const [enableDoclingChunk, setEnableDoclingChunk] = useState(true); @@ -45,8 +43,7 @@ export function IndexForm({ onClose, onIndexed }: Props) { latechunk: enableLateChunk, doclingChunk: enableDoclingChunk, chunkSize: chunkSize, - chunkOverlap: chunkOverlap, - retrievalMode: retrievalMode==='fts' ? 'bm25' : retrievalMode, + retrievalMode: retrievalMode, windowSize: windowSize, enableEnrich: enableEnrich, embeddingModel: embeddingModel, @@ -108,8 +105,8 @@ export function IndexForm({ onClose, onIndexed }: Props) { <div> <label className="flex items-center gap-1 text-xs uppercase tracking-wide text-gray-300 mb-1">Retrieval mode <InfoTooltip text="Choose how chunks are found. Hybrid combines full-text search with vectors; FTS uses textual matching only; Vector relies purely on dense similarity." /></label> <div className="flex gap-3"> - {(['hybrid','vector','fts'] as const).map((m)=>( - <button key={m} onClick={()=>setRetrievalMode(m)} className={`px-3 py-1 rounded text-xs font-sans ${retrievalMode===m?'bg-white/20':'bg-white/10 hover:bg-white/20'}`}>{m==='fts' ? 'FTS' : m}</button> + {([['hybrid','hybrid'],['vector_only','vector'],['fts_only','FTS']] as const).map(([m,label])=>( + <button key={m} onClick={()=>setRetrievalMode(m)} className={`px-3 py-1 rounded text-xs font-sans ${retrievalMode===m?'bg-white/20':'bg-white/10 hover:bg-white/20'}`}>{label}</button> ))} </div> <div className="grid grid-cols-2 gap-4 mt-3"> @@ -127,14 +124,6 @@ export function IndexForm({ onClose, onIndexed }: Props) { <label className="flex items-center gap-1 text-xs mb-1 text-gray-400">Chunk size <InfoTooltip text="Maximum token length for each chunk. Both legacy and high-recall modes now use token-based sizing." size={12} /></label> <GlassInput type="number" value={chunkSize} onChange={(e) => setChunkSize(parseInt(e.target.value))} /> </div> - <div> - <label className="flex items-center gap-1 text-xs mb-1 text-gray-400">Chunk overlap <InfoTooltip text="Tokens reused between adjacent chunks to preserve context." size={12} /></label> - <GlassInput - type="number" - value={chunkOverlap} - onChange={(e) => setChunkOverlap(parseInt(e.target.value))} - /> - </div> </div> {/* Embedding & Overview models */} diff --git a/src/components/IndexWizard.tsx b/src/components/IndexWizard.tsx deleted file mode 100644 index 8e14a9dd..00000000 --- a/src/components/IndexWizard.tsx +++ /dev/null @@ -1,72 +0,0 @@ -"use client"; -import { useState } from 'react'; -import { ModelSelect } from '@/components/ModelSelect'; - -interface Props { - onClose: () => void; -} - -export function IndexWizard({ onClose }: Props) { - const [files, setFiles] = useState<FileList | null>(null); - const [chunkSize, setChunkSize] = useState(512); - const [chunkOverlap, setChunkOverlap] = useState(64); - const [embeddingModel, setEmbeddingModel] = useState<string>(); - // TODO: more params - - const handleFile = (e: React.ChangeEvent<HTMLInputElement>) => { - setFiles(e.target.files); - }; - - return ( - <div className="fixed inset-0 bg-black/60 backdrop-blur flex items-center justify-center z-50"> - <div className="bg-gray-900 w-[600px] max-h-[90vh] overflow-auto rounded-xl p-6 text-white space-y-6"> - <h2 className="text-lg font-semibold">Create new index</h2> - - <div className="space-y-4"> - <div> - <label className="block text-sm mb-1">Document files</label> - <input type="file" accept="application/pdf,.docx,.doc,.html,.htm,.md,.txt" multiple onChange={handleFile} className="text-sm" /> - </div> - - <div className="grid grid-cols-2 gap-4"> - <div> - <label className="block text-sm mb-1">Chunk size</label> - <input - type="number" - value={chunkSize} - onChange={(e) => setChunkSize(parseInt(e.target.value))} - className="w-full bg-gray-800 rounded px-2 py-1" - /> - </div> - <div> - <label className="block text-sm mb-1">Chunk overlap</label> - <input - type="number" - value={chunkOverlap} - onChange={(e) => setChunkOverlap(parseInt(e.target.value))} - className="w-full bg-gray-800 rounded px-2 py-1" - /> - </div> - </div> - - <div> - <label className="block text-sm mb-1">Embedding model</label> - <ModelSelect type="embedding" value={embeddingModel} onChange={setEmbeddingModel} /> - </div> - </div> - - <div className="flex justify-end gap-3 pt-4 border-t border-white/10"> - <button onClick={onClose} className="px-4 py-2 bg-gray-700 rounded hover:bg-gray-600 text-sm"> - Cancel - </button> - <button - disabled={!files || !embeddingModel} - className="px-4 py-2 bg-green-600 rounded disabled:opacity-40 text-sm" - > - Start indexing - </button> - </div> - </div> - </div> - ); -} \ No newline at end of file diff --git a/src/components/Markdown.tsx b/src/components/Markdown.tsx index 13285251..d3c86637 100644 --- a/src/components/Markdown.tsx +++ b/src/components/Markdown.tsx @@ -1,13 +1,12 @@ -// eslint-disable-next-line @typescript-eslint/ban-ts-comment -// @ts-nocheck 'use client' import dynamic from 'next/dynamic' import React, { useMemo } from 'react' import remarkGfm from 'remark-gfm' +import remarkBreaks from 'remark-breaks' // Dynamically import react-markdown to avoid SSR issues -const ReactMarkdown: any = dynamic(() => import('react-markdown') as any, { ssr: false }) +const ReactMarkdown = dynamic(() => import('react-markdown'), { ssr: false }) interface MarkdownProps { text: string @@ -15,10 +14,13 @@ interface MarkdownProps { } export default function Markdown({ text, className = '' }: MarkdownProps) { - const plugins = useMemo(() => [remarkGfm], []) + // remark-breaks renders single newlines as <br>: model answers use them + // as intentional line breaks, and without the plugin markdown collapses + // them into spaces. This replaces the old whitespace-pre-wrap approach, + // which rendered every newline TWICE (markdown paragraph + literal break). + const plugins = useMemo(() => [remarkGfm, remarkBreaks], []) return ( - <div className={`prose prose-invert max-w-none ${className}`}> - {/* @ts-ignore โ€“ react-markdown type doesn't recognise remarkPlugins array */} + <div className={`prose prose-invert max-w-none prose-p:my-2 prose-headings:mt-3 prose-headings:mb-2 prose-ul:my-2 prose-ol:my-2 prose-pre:my-2 prose-hr:my-3 ${className}`}> <ReactMarkdown remarkPlugins={plugins} components={{ @@ -31,4 +33,4 @@ export default function Markdown({ text, className = '' }: MarkdownProps) { </ReactMarkdown> </div> ) -} \ No newline at end of file +} diff --git a/src/components/ModelSelect.tsx b/src/components/ModelSelect.tsx index 28e4f410..b90cdbac 100644 --- a/src/components/ModelSelect.tsx +++ b/src/components/ModelSelect.tsx @@ -1,5 +1,5 @@ import { useEffect, useState } from 'react'; -import { chatAPI, ModelsResponse } from '@/lib/api'; +import { chatAPI, ModelsResponse, DEFAULT_GENERATION_MODEL, DEFAULT_EMBEDDING_MODEL } from '@/lib/api'; interface Props { value: string | undefined; @@ -22,9 +22,10 @@ export function ModelSelect({ value, onChange, type, className, placeholder }: P if (!mounted) return; const list = type === 'generation' ? res.generation_models : res.embedding_models; setModels(list); - // Auto-select default qwen3:0.6b if available and not chosen yet - if(!value && list.includes('qwen3:0.6b')){ - onChange('qwen3:0.6b'); + // Auto-select the role default when available and nothing is chosen yet + const preferred = type === 'generation' ? DEFAULT_GENERATION_MODEL : DEFAULT_EMBEDDING_MODEL; + if(!value && list.includes(preferred)){ + onChange(preferred); } setLoading(false); }) diff --git a/src/components/SessionIndexInfo.tsx b/src/components/SessionIndexInfo.tsx index 3c9098dc..543e1a5e 100644 --- a/src/components/SessionIndexInfo.tsx +++ b/src/components/SessionIndexInfo.tsx @@ -17,11 +17,12 @@ export default function SessionIndexInfo({ sessionId, onClose }: Props) { (async () => { try { const data = await chatAPI.getSessionIndexes(sessionId); - const first = data.indexes[0]; - if(first){ - setSession(first.session??{...first, title:first.name, model_used:first.model_used||''}); - setFiles(first.documents?.map((d:any)=>d.filename) || []); - setIndexMeta(first.metadata || {}); + // Chat queries run against the LAST linked index โ€” show that one. + const last = data.indexes[data.indexes.length - 1]; + if(last){ + setSession(last.session??{...last, title:last.name, model_used:last.model_used||''}); + setFiles(last.documents?.map((d:any)=>d.filename) || []); + setIndexMeta(last.metadata || {}); } else { setError('No indexes linked to this chat'); } @@ -77,8 +78,7 @@ export default function SessionIndexInfo({ sessionId, onClose }: Props) { if (indexStatus === 'functional') { // Check if we have complete configuration metadata - const hasCompleteConfig = indexMeta.chunk_size && - indexMeta.chunk_overlap !== undefined && + const hasCompleteConfig = indexMeta.chunk_size && indexMeta.retrieval_mode && indexMeta.embedding_model; @@ -185,12 +185,6 @@ export default function SessionIndexInfo({ sessionId, onClose }: Props) { </p> </div> )} - {typeof indexMeta.chunk_overlap==='number' && ( - <div> - <span className="block text-xs uppercase tracking-wide text-gray-300 mb-1">Chunk overlap</span> - <p className="text-sm">{indexMeta.chunk_overlap} tokens</p> - </div> - )} </div> {/* Context and Features */} diff --git a/src/components/demo.tsx b/src/components/demo.tsx index 970bf180..14d0f0eb 100644 --- a/src/components/demo.tsx +++ b/src/components/demo.tsx @@ -1,45 +1,47 @@ "use client"; import { useState, useEffect } from "react" -import { LocalGPTChat } from "@/components/ui/localgpt-chat" import { SessionSidebar } from "@/components/ui/session-sidebar" import { SessionChat } from '@/components/ui/session-chat' import { chatAPI, ChatSession } from "@/lib/api" import { LandingMenu } from "@/components/LandingMenu"; import { IndexForm } from "@/components/IndexForm"; -import SessionIndexInfo from "@/components/SessionIndexInfo"; import IndexPicker from "@/components/IndexPicker"; import { QuickChat } from '@/components/ui/quick-chat' export function Demo() { const [currentSessionId, setCurrentSessionId] = useState<string | undefined>() - const [currentSession, setCurrentSession] = useState<ChatSession | null>(null) const [showConversation, setShowConversation] = useState(false) const [backendStatus, setBackendStatus] = useState<'checking' | 'connected' | 'error'>('checking') const [sidebarRef, setSidebarRef] = useState<{ refreshSessions: () => Promise<void> } | null>(null) const [homeMode, setHomeMode] = useState<'HOME' | 'INDEX' | 'CHAT_EXISTING' | 'QUICK_CHAT'>('HOME') - const [showIndexInfo, setShowIndexInfo] = useState(false) const [showIndexPicker, setShowIndexPicker] = useState(false) const [sidebarOpen, setSidebarOpen] = useState(true) - console.log('Demo component rendering...') - useEffect(() => { console.log('Demo component mounted') + let stopped = false + // Poll until the backend answers so a backend that starts after the + // frontend still clears the "Backend offline" banner. + let interval: ReturnType<typeof setInterval> | null = null + const checkBackendHealth = async () => { + try { + const health = await chatAPI.checkHealth() + if (stopped) return + setBackendStatus('connected') + console.log('Backend connected:', health) + if (interval) clearInterval(interval) + } catch (error) { + if (stopped) return + console.error('Backend health check failed:', error) + setBackendStatus('error') + } + } checkBackendHealth() + interval = setInterval(checkBackendHealth, 5000) + return () => { stopped = true; if (interval) clearInterval(interval) } }, []) - const checkBackendHealth = async () => { - try { - const health = await chatAPI.checkHealth() - setBackendStatus('connected') - console.log('Backend connected:', health) - } catch (error) { - console.error('Backend health check failed:', error) - setBackendStatus('error') - } - } - const handleSessionSelect = (sessionId: string) => { setCurrentSessionId(sessionId) setShowConversation(true) @@ -49,14 +51,11 @@ export function Demo() { const handleNewSession = () => { // Reset state and return to landing page so user can choose chat type setCurrentSessionId(undefined) - setCurrentSession(null) setShowConversation(false) // Hide conversation view & sidebar setHomeMode('HOME') // Show landing selector (Create index / Chat with index / LLM Chat) } const handleSessionChange = async (session: ChatSession) => { - setCurrentSession(session) - // Update the current session ID if it changed (e.g., brand-new session) if (session.id !== currentSessionId) { setCurrentSessionId(session.id) @@ -72,16 +71,6 @@ export function Demo() { if (currentSessionId === deletedSessionId) { // Stay in conversation mode but show empty state setCurrentSessionId(undefined) - setCurrentSession(null) - } - } - - const handleStartConversation = () => { - if (backendStatus === 'connected') { - // Just show empty state, don't create session yet - handleNewSession() - } else { - setShowConversation(true) } } @@ -163,23 +152,27 @@ export function Demo() { </main> {homeMode==='INDEX' && ( - <div className="fixed inset-0 flex items-center justify-center bg-black/50 backdrop-blur-sm z-50"> - <IndexForm onClose={()=>setHomeMode('HOME')} onIndexed={(s)=>{setHomeMode('CHAT_EXISTING'); handleSessionSelect(s.id);}} /> + <div className="fixed inset-0 flex items-center justify-center bg-black/50 backdrop-blur-sm z-50 p-4"> + <div className="max-h-[92vh] overflow-y-auto"> + <IndexForm onClose={()=>setHomeMode('HOME')} onIndexed={(s)=>{setHomeMode('CHAT_EXISTING'); handleSessionSelect(s.id);}} /> + </div> </div> )} - {showIndexInfo && currentSessionId && ( - <SessionIndexInfo sessionId={currentSessionId} onClose={()=>setShowIndexInfo(false)} /> - )} - {showIndexPicker && ( <IndexPicker onClose={()=>setShowIndexPicker(false)} onSelect={async (idxId)=>{ // create session and link index then open chat - const session = await chatAPI.createSession() - await chatAPI.linkIndexToSession(session.id, idxId) - setShowIndexPicker(false) - setHomeMode('CHAT_EXISTING') - handleSessionSelect(session.id) + try { + const session = await chatAPI.createSession() + await chatAPI.linkIndexToSession(session.id, idxId) + setShowIndexPicker(false) + setHomeMode('CHAT_EXISTING') + handleSessionSelect(session.id) + } catch (error) { + console.error('Failed to start chat with index:', error) + setShowIndexPicker(false) + alert('Could not start a chat with that index. Is the backend running?') + } }} /> )} </div> diff --git a/src/components/ui/GlassSelect.tsx b/src/components/ui/GlassSelect.tsx deleted file mode 100644 index 940975ad..00000000 --- a/src/components/ui/GlassSelect.tsx +++ /dev/null @@ -1,13 +0,0 @@ -"use client"; -import React, { SelectHTMLAttributes } from 'react'; - -export function GlassSelect(props: SelectHTMLAttributes<HTMLSelectElement>) { - return ( - <select - {...props} - className={`w-full rounded bg-white/5 hover:bg-white/10 focus:bg-white/10 px-2 py-1 text-sm font-sans text-white outline-none focus:ring-2 focus:ring-white/20 transition ${props.className || ''}`} - > - {props.children} - </select> - ); -} \ No newline at end of file diff --git a/src/components/ui/badge.tsx b/src/components/ui/badge.tsx deleted file mode 100644 index 02054139..00000000 --- a/src/components/ui/badge.tsx +++ /dev/null @@ -1,46 +0,0 @@ -import * as React from "react" -import { Slot } from "@radix-ui/react-slot" -import { cva, type VariantProps } from "class-variance-authority" - -import { cn } from "@/lib/utils" - -const badgeVariants = cva( - "inline-flex items-center justify-center rounded-md border px-2 py-0.5 text-xs font-medium w-fit whitespace-nowrap shrink-0 [&>svg]:size-3 gap-1 [&>svg]:pointer-events-none focus-visible:border-ring focus-visible:ring-ring/50 focus-visible:ring-[3px] aria-invalid:ring-destructive/20 dark:aria-invalid:ring-destructive/40 aria-invalid:border-destructive transition-[color,box-shadow] overflow-hidden", - { - variants: { - variant: { - default: - "border-transparent bg-primary text-primary-foreground [a&]:hover:bg-primary/90", - secondary: - "border-transparent bg-secondary text-secondary-foreground [a&]:hover:bg-secondary/90", - destructive: - "border-transparent bg-destructive text-white [a&]:hover:bg-destructive/90 focus-visible:ring-destructive/20 dark:focus-visible:ring-destructive/40 dark:bg-destructive/60", - outline: - "text-foreground [a&]:hover:bg-accent [a&]:hover:text-accent-foreground", - }, - }, - defaultVariants: { - variant: "default", - }, - } -) - -function Badge({ - className, - variant, - asChild = false, - ...props -}: React.ComponentProps<"span"> & - VariantProps<typeof badgeVariants> & { asChild?: boolean }) { - const Comp = asChild ? Slot : "span" - - return ( - <Comp - data-slot="badge" - className={cn(badgeVariants({ variant }), className)} - {...props} - /> - ) -} - -export { Badge, badgeVariants } diff --git a/src/components/ui/chat-bubble-demo.tsx b/src/components/ui/chat-bubble-demo.tsx deleted file mode 100644 index 032b63f8..00000000 --- a/src/components/ui/chat-bubble-demo.tsx +++ /dev/null @@ -1,103 +0,0 @@ -"use client" - -import { - ChatBubble, - ChatBubbleAvatar, - ChatBubbleMessage -} from "@/components/ui/chat-bubble" -import { Copy, RefreshCcw } from "lucide-react" - -const messages = [ - { - id: 1, - message: "Help me with my essay.", - sender: "user", - }, - { - id: 2, - message: "I can help you with that. What do you need help with?", - sender: "bot", - }, -] - -const actionIcons = [ - { icon: Copy, type: "Copy" }, - { icon: RefreshCcw, type: "Regenerate" }, -] - -export function ChatBubbleVariants() { - return ( - <div className="max-w-md space-y-4 p-4"> - <ChatBubble variant="sent"> - <ChatBubbleAvatar fallback="US" src="https://images.unsplash.com/photo-1534528741775-53994a69daeb?w=64&h=64&q=80&crop=faces&fit=crop" /> - <ChatBubbleMessage variant="sent"> - I have a question about the library. - </ChatBubbleMessage> - </ChatBubble> - - <ChatBubble variant="received"> - <ChatBubbleAvatar fallback="AI" src="https://images.unsplash.com/photo-1677442136019-21780ecad995?w=64&h=64&q=80&crop=faces&fit=crop" /> - <ChatBubbleMessage> - Sure, I'd be happy to help! - </ChatBubbleMessage> - </ChatBubble> - </div> - ) -} - -export function ChatBubbleAiLayout() { - return ( - <div className="max-w-md divide-y"> - {messages.map((message, index) => { - const variant = message.sender === "user" ? "sent" : "received" - return ( - <div key={message.id} className="py-6 first:pt-0 last:pb-0"> - <div className="flex gap-3"> - <ChatBubbleAvatar - src={variant === "sent" - ? "https://images.unsplash.com/photo-1534528741775-53994a69daeb?w=64&h=64&q=80&crop=faces&fit=crop" - : "https://images.unsplash.com/photo-1677442136019-21780ecad995?w=64&h=64&q=80&crop=faces&fit=crop" - } - fallback={variant === "sent" ? "US" : "L"} - /> - <div className="flex-1"> - {message.message} - {message.sender === "bot" && ( - <div className="flex gap-2 mt-2"> - {actionIcons.map(({ icon: Icon, type }) => ( - <button - key={type} - onClick={() => console.log(`Action ${type} clicked for message ${index}`)} - className="p-1 hover:bg-muted rounded-md transition-colors" - > - <Icon className="size-3" /> - </button> - ))} - </div> - )} - </div> - </div> - </div> - ) - })} - </div> - ) -} - -export function ChatBubbleStates() { - return ( - <div className="max-w-md space-y-4 p-4"> - <ChatBubble variant="received"> - <ChatBubbleAvatar fallback="L" /> - <ChatBubbleMessage isLoading /> - </ChatBubble> - - <ChatBubble variant="received"> - <ChatBubbleAvatar fallback="L" /> - <ChatBubbleMessage className="bg-destructive/10 text-destructive"> - Error processing request - </ChatBubbleMessage> - </ChatBubble> - </div> - ) -} \ No newline at end of file diff --git a/src/components/ui/chat-bubble.tsx b/src/components/ui/chat-bubble.tsx index 87718f4a..97fcdc83 100644 --- a/src/components/ui/chat-bubble.tsx +++ b/src/components/ui/chat-bubble.tsx @@ -1,68 +1,7 @@ "use client" -import * as React from "react" import { cn } from "@/lib/utils" import { Avatar, AvatarFallback, AvatarImage } from "@/components/ui/avatar" -import { Button } from "@/components/ui/button" -import { MessageLoading } from "@/components/ui/message-loading"; - -interface ChatBubbleProps { - variant?: "sent" | "received" - layout?: "default" | "ai" - className?: string - children: React.ReactNode -} - -export function ChatBubble({ - variant = "received", - layout = "default", // eslint-disable-line @typescript-eslint/no-unused-vars - className, - children, -}: ChatBubbleProps) { - return ( - <div - className={cn( - "flex items-start gap-2 mb-4", - variant === "sent" && "flex-row-reverse", - className, - )} - > - {children} - </div> - ) -} - -interface ChatBubbleMessageProps { - variant?: "sent" | "received" - isLoading?: boolean - className?: string - children?: React.ReactNode -} - -export function ChatBubbleMessage({ - variant = "received", - isLoading, - className, - children, -}: ChatBubbleMessageProps) { - return ( - <div - className={cn( - "rounded-lg p-3", - variant === "sent" ? "bg-primary text-primary-foreground" : "bg-muted", - className - )} - > - {isLoading ? ( - <div className="flex items-center space-x-2"> - <MessageLoading /> - </div> - ) : ( - children - )} - </div> - ) -} interface ChatBubbleAvatarProps { src?: string @@ -78,44 +17,11 @@ export function ChatBubbleAvatar({ return ( <Avatar className={cn("h-8 w-8", className)}> {src && <AvatarImage src={src} />} - <AvatarFallback>{fallback}</AvatarFallback> + {/* Explicit light scheme: the shadcn default (bg-muted + text-black) is + invisible in this app's dark theme once there is no image on top. */} + <AvatarFallback className="bg-white text-black text-xs font-medium"> + {fallback} + </AvatarFallback> </Avatar> ) } - -interface ChatBubbleActionProps { - icon?: React.ReactNode - onClick?: () => void - className?: string -} - -export function ChatBubbleAction({ - icon, - onClick, - className, -}: ChatBubbleActionProps) { - return ( - <Button - variant="ghost" - size="icon" - className={cn("h-6 w-6", className)} - onClick={onClick} - > - {icon} - </Button> - ) -} - -export function ChatBubbleActionWrapper({ - className, - children, -}: { - className?: string - children: React.ReactNode -}) { - return ( - <div className={cn("flex items-center gap-1 mt-2", className)}> - {children} - </div> - ) -} \ No newline at end of file diff --git a/src/components/ui/chat-input.tsx b/src/components/ui/chat-input.tsx index a6ff6770..1012ca36 100644 --- a/src/components/ui/chat-input.tsx +++ b/src/components/ui/chat-input.tsx @@ -2,9 +2,10 @@ import * as React from "react" import { useState, useRef } from "react" -import { ArrowUp, Settings as SettingsIcon, Plus, X, FileText } from "lucide-react" +import { ArrowUp, Settings as SettingsIcon, X, FileText, Paperclip } from "lucide-react" import { Button } from "@/components/ui/button" import { AttachedFile } from "@/lib/types" +import { generateUUID } from "@/lib/api" interface ChatInputProps { onSendMessage: (message: string, attachedFiles?: AttachedFile[]) => Promise<void> @@ -103,7 +104,7 @@ export function ChatInput({ file.name.toLowerCase().endsWith('.md') || file.name.toLowerCase().endsWith('.txt')) { newFiles.push({ - id: crypto.randomUUID(), + id: generateUUID(), name: file.name, size: file.size, type: file.type, @@ -163,7 +164,7 @@ export function ChatInput({ )} <div className="bg-white/5 backdrop-blur border border-white/10 rounded-2xl px-5 pt-4 pb-3 space-y-2"> - {/* Hidden file input (kept for future use) */} + {/* Hidden file input, opened via the attach button below */} <input ref={fileInputRef} type="file" accept=".pdf,.docx,.doc,.html,.htm,.md,.txt" multiple onChange={handleFileChange} className="hidden" /> {/* Textarea */} @@ -182,6 +183,15 @@ export function ChatInput({ {/* Action row */} <div className="mt-1 flex items-center justify-between"> <div className="flex items-center gap-4"> + <button + type="button" + onClick={handleFileAttach} + disabled={disabled || isLoading} + className="flex items-center gap-1 p-2 text-gray-400 hover:text-white hover:bg-gray-800 rounded-full transition-colors disabled:opacity-50 disabled:cursor-not-allowed" + title="Attach files" + > + <Paperclip className="w-5 h-5" /> + </button> <button type="button" onClick={()=>onOpenSettings && onOpenSettings()} diff --git a/src/components/ui/chat-settings-modal.tsx b/src/components/ui/chat-settings-modal.tsx index 3e9d9f69..c530c0c6 100644 --- a/src/components/ui/chat-settings-modal.tsx +++ b/src/components/ui/chat-settings-modal.tsx @@ -1,5 +1,6 @@ "use client"; +import { useEffect } from 'react'; import { GlassToggle } from '@/components/ui/GlassToggle'; import { InfoTooltip } from '@/components/ui/InfoTooltip'; @@ -43,7 +44,7 @@ const optionHelp: Record<string,string> = { 'RAG (no-triage)':'Force retrieval on every query; disables index-selection triage.', 'Verify answer':'Runs an extra LLM pass to self-critique the draft answer.', 'Streaming':'Send tokens to the UI as they are generated.', - 'AI reranker':'Re-orders retrieved chunks with a cross-encoder (higher quality, more latency).', + 'AI reranker':'On by default (eval arm G). Re-orders retrieved chunks with Qwen3-Reranker-4B: measurably better ranking, but it loads 7.5GB of weights and added ~12.7s per query in our eval.', 'Expand context window':'Adds neighbour chunks around each top chunk to provide more context.', 'Context window size':'How many neighbour chunks to include on each side.', 'Retrieval chunks':'Number of chunks fetched before reranking.', @@ -53,6 +54,15 @@ const optionHelp: Record<string,string> = { }; export function ChatSettingsModal({ options, onClose }: Props) { + // Escape closes the dialog + useEffect(() => { + const onKeyDown = (e: KeyboardEvent) => { + if (e.key === 'Escape') onClose(); + }; + window.addEventListener('keydown', onKeyDown); + return () => window.removeEventListener('keydown', onKeyDown); + }, [onClose]); + const renderOption = (opt: SettingOption) => { switch (opt.type) { case 'toggle': @@ -82,7 +92,7 @@ export function ChatSettingsModal({ options, onClose }: Props) { step={opt.step || 1} value={opt.value} onChange={(e) => opt.setter(Number(e.target.value))} - className="w-full h-2 bg-gray-700 rounded-lg appearance-none cursor-pointer slider" + className="w-full h-2 bg-gray-700 rounded-lg appearance-none cursor-pointer" style={{ background: `linear-gradient(to right, #3b82f6 0%, #3b82f6 ${((opt.value - opt.min) / (opt.max - opt.min)) * 100}%, #374151 ${((opt.value - opt.min) / (opt.max - opt.min)) * 100}%, #374151 100%)` }} @@ -146,7 +156,7 @@ export function ChatSettingsModal({ options, onClose }: Props) { return ( <div className="fixed inset-0 bg-black/60 backdrop-blur-sm flex items-center justify-center z-50 p-4"> - <div className="bg-white/5 backdrop-blur rounded-xl w-full max-w-xl max-h-full overflow-y-auto p-6 text-white space-y-6"> + <div role="dialog" aria-modal="true" aria-label="Chat settings" className="bg-white/5 backdrop-blur rounded-xl w-full max-w-xl max-h-full overflow-y-auto p-6 text-white space-y-6"> <h2 className="text-lg font-semibold mb-6">Chat Settings</h2> <div className="space-y-6"> diff --git a/src/components/ui/conversation-page.tsx b/src/components/ui/conversation-page.tsx index 191f1554..b62db831 100644 --- a/src/components/ui/conversation-page.tsx +++ b/src/components/ui/conversation-page.tsx @@ -1,14 +1,13 @@ "use client" import * as React from "react" -import { useRef, useEffect, useState } from "react" +import { useRef, useEffect, useState, useCallback } from "react" import { ChatBubbleAvatar, } from "@/components/ui/chat-bubble" import { Copy, RefreshCcw, ThumbsUp, ThumbsDown, Volume2, MoreHorizontal, ChevronDown, Loader2, CheckCircle, XOctagon } from "lucide-react" import { ScrollArea } from "@/components/ui/scroll-area" import { ChatMessage } from "@/lib/api" -import { cn } from "@/lib/utils" import Markdown from "@/components/Markdown" import { normalizeWhitespace } from "@/utils/textNormalization" @@ -41,8 +40,12 @@ function Citation({doc, idx}: {doc:any, idx:number}){ // NEW: Expandable list of citations per assistant message function CitationsBlock({docs}:{docs:any[]}){ - const scored = docs.filter(d => d.rerank_score || d.score || d._distance) - scored.sort((a, b) => (b.rerank_score ?? b.score ?? 1/b._distance) - (a.rerank_score ?? a.score ?? 1/a._distance)) + // != null (not truthiness) so a perfect-match _distance of exactly 0 keeps the doc + const scored = docs.filter(d => d.rerank_score != null || d.score != null || d._distance != null) + // Total-order sort key โ€” rerank_score > score > inverted distance (smaller + // distance = better). The epsilon keeps _distance 0 finite (no Infinity/NaN). + const rankOf = (d: any) => d.rerank_score ?? d.score ?? (d._distance != null ? 1 / (d._distance + 1e-9) : 0) + scored.sort((a, b) => rankOf(b) - rankOf(a)) const [expanded, setExpanded] = useState(false); if (scored.length === 0) return null; @@ -111,7 +114,7 @@ function ThinkingText({ text }: { text: string }) { </details> )} {visibleText.trim() && ( - <Markdown text={normalizeWhitespace(visibleText)} className="whitespace-pre-wrap" /> + <Markdown text={normalizeWhitespace(visibleText)} className="break-words" /> )} </> ); @@ -140,7 +143,7 @@ function StructuredMessageBlock({ content }: { content: Array<Record<string, any {visibleSteps.map((step: any, index: number) => { if (step.key && step.label) { const borderCls = statusBorder[step.status] || statusBorder['pending'] - const statusClass = `timeline-card card my-1 py-2 pl-3 pr-2 bg-[#0d0d0d] rounded border-l-2 ${borderCls}` + const statusClass = `my-1 py-2 pl-3 pr-2 bg-[#0d0d0d] rounded border-l-2 ${borderCls}` return ( <div key={step.key} className={statusClass}> @@ -151,7 +154,7 @@ function StructuredMessageBlock({ content }: { content: Array<Record<string, any {/* Details for each step */} {step.key === 'final' && step.details && typeof step.details === 'object' && !Array.isArray(step.details) ? ( <div className="space-y-3"> - <div className="whitespace-pre-wrap text-gray-100"> + <div className="break-words text-gray-100"> <ThinkingText text={normalizeWhitespace(step.details.answer)} /> </div> {!hasSubAnswers && step.details.source_documents && step.details.source_documents.length > 0 && ( @@ -159,7 +162,7 @@ function StructuredMessageBlock({ content }: { content: Array<Record<string, any )} </div> ) : step.key === 'final' && step.details && typeof step.details === 'string' ? ( - <div className="whitespace-pre-wrap text-gray-100"> + <div className="break-words text-gray-100"> <ThinkingText text={normalizeWhitespace(step.details)} /> </div> ) : Array.isArray(step.details) ? ( @@ -197,6 +200,108 @@ function StructuredMessageBlock({ content }: { content: Array<Record<string, any ); } +// Normalize the various message content shapes to plain text (copy/paste, +// action callbacks). String โ†’ as-is; array โ†’ text/answer parts joined with a +// real newline; { steps } โ†’ the final answer text. +function messageContentToText(content: ChatMessage['content']): string { + if (typeof content === 'string') return content; + if (Array.isArray(content)) { + return content.map((s: any) => s?.text || s?.answer || '').filter(Boolean).join('\n'); + } + if (content && typeof content === 'object' && Array.isArray((content as any).steps)) { + const steps = (content as any).steps as any[]; + const finalStep = steps.find((s: any) => s.key === 'final') ?? steps[steps.length - 1]; + const details = finalStep?.details; + if (typeof details === 'string') return details; + if (details && typeof details === 'object' && typeof details.answer === 'string') return details.answer; + return ''; + } + return ''; +} + +interface MessageBubbleProps { + message: ChatMessage + onAction: (action: string, messageId: string, messageContent: ChatMessage['content']) => void +} + +// Memoized: during streaming, only the bubble whose message object actually +// changed re-renders โ€” previously-rendered messages no longer re-parse their +// Markdown on every token. +const MessageBubble = React.memo(function MessageBubble({ message, onAction }: MessageBubbleProps) { + const isUser = message.sender === "user" + + return ( + <div className="w-full group"> + <div className={`flex gap-3 ${isUser ? 'justify-end' : 'justify-start'}`}> + {!isUser && ( + <ChatBubbleAvatar + fallback="AI" + className="mt-1 flex-shrink-0 text-black" + /> + )} + + <div className={`flex flex-col space-y-2 ${isUser ? 'items-end' : 'items-start'} max-w-full md:max-w-3xl min-w-0`}> + <div + className={`rounded-2xl px-5 py-4 ${ + isUser + ? "bg-white text-black" + : "bg-gray-800 text-gray-100" + }`} + > + {message.isLoading ? ( + <div className="flex items-center space-x-2"> + <div className="flex space-x-1"> + <div className="w-2 h-2 bg-gray-400 rounded-full animate-bounce"></div> + <div className="w-2 h-2 bg-gray-400 rounded-full animate-bounce" style={{animationDelay: '0.1s'}}></div> + <div className="w-2 h-2 bg-gray-400 rounded-full animate-bounce" style={{animationDelay: '0.2s'}}></div> + </div> + </div> + ) : ( + <div className="break-words text-base leading-relaxed"> + {typeof message.content === 'string' + ? <ThinkingText text={normalizeWhitespace(message.content)} /> + : <StructuredMessageBlock content={message.content} /> + } + </div> + )} + </div> + + {!isUser && !message.isLoading && ( + <div className="flex items-center gap-1 opacity-0 group-hover:opacity-100 transition-opacity duration-200"> + {actionIcons.map(({ icon: Icon, type, action }) => ( + <button + key={action} + onClick={() => onAction(action, message.id, message.content)} + className="p-1.5 hover:bg-gray-700 rounded-md transition-colors text-gray-400 hover:text-gray-200" + title={type} + > + <Icon className="w-3.5 h-3.5" /> + </button> + ))} + </div> + )} + + {/* Global citations only for plain-string messages */} + {(!isUser && + !message.isLoading && + typeof message.content === 'string' && + Array.isArray((message as any).metadata?.source_documents) && + (message as any).metadata.source_documents.length > 0) && ( + <CitationsBlock docs={(message as any).metadata.source_documents} /> + )} + </div> + + {isUser && ( + <ChatBubbleAvatar + className="mt-1 flex-shrink-0 text-black" + fallback="U" + /> + )} + </div> + </div> + ) +}) + export function ConversationPage({ messages, isLoading = false, @@ -250,29 +355,27 @@ export function ConversationPage({ }, 100) } - const handleAction = (action: string, messageId: string, messageContent: string) => { - if (onAction) { - // For structured messages, we'll just join the text parts for copy/paste - let contentToPass: string; - if (typeof messageContent === 'string') { - contentToPass = messageContent; - } else if (Array.isArray(messageContent)) { - contentToPass = (messageContent as any[]).map((s: any) => s.text || s.answer || '').join('\n'); - } else if (messageContent && typeof messageContent === 'object' && Array.isArray((messageContent as any).steps)) { - // For {steps: Step[]} structure - contentToPass = (messageContent as any).steps.map((s: any) => s.label + (s.details ? (typeof s.details === 'string' ? (': ' + s.details) : '') : '')).join('\n'); - } else { - contentToPass = ''; - } - onAction(action, messageId, contentToPass) + // Keep the latest onAction in a ref so handleAction below stays referentially + // stable (memoized bubbles would otherwise re-render on every parent render) + // while always dispatching to the freshest handler. + const onActionRef = useRef(onAction) + useEffect(() => { + onActionRef.current = onAction + }, [onAction]) + + const handleAction = useCallback((action: string, messageId: string, messageContent: ChatMessage['content']) => { + const contentToPass = messageContentToText(messageContent) + const handler = onActionRef.current + if (handler) { + handler(action, messageId, contentToPass) return } - + console.log(`Action ${action} clicked for message ${messageId}`) // Handle different actions here switch (action) { case 'copy': - navigator.clipboard.writeText(messageContent) + navigator.clipboard.writeText(contentToPass) break case 'regenerate': // Regenerate AI response @@ -290,90 +393,15 @@ export function ConversationPage({ // Show more options break } - } + }, []) return ( <div className={`flex flex-col h-full bg-black relative overflow-hidden ${className}`}> <ScrollArea ref={scrollAreaRef} className="flex-1 h-full px-4 pt-4 pb-6 min-h-0"> <div className="max-w-4xl mx-auto space-y-6"> - {messages.map((message) => { - const isUser = message.sender === "user" - - return ( - <div key={message.id} className="w-full group"> - <div className={`flex gap-3 ${isUser ? 'justify-end' : 'justify-start'}`}> - {!isUser && ( - <ChatBubbleAvatar - fallback="AI" - className="mt-1 flex-shrink-0 text-black" - /> - )} - - <div className={`flex flex-col space-y-2 ${isUser ? 'items-end' : 'items-start'} max-w-full md:max-w-3xl`}> - <div - className={`rounded-2xl px-5 py-4 ${ - isUser - ? "bg-white text-black" - : "bg-gray-800 text-gray-100" - }`} - > - {message.isLoading ? ( - <div className="flex items-center space-x-2"> - <div className="flex space-x-1"> - <div className="w-2 h-2 bg-gray-400 rounded-full animate-bounce"></div> - <div className="w-2 h-2 bg-gray-400 rounded-full animate-bounce" style={{animationDelay: '0.1s'}}></div> - <div className="w-2 h-2 bg-gray-400 rounded-full animate-bounce" style={{animationDelay: '0.2s'}}></div> - </div> - </div> - ) : ( - <div className="whitespace-pre-wrap text-base leading-relaxed"> - {typeof message.content === 'string' - ? <ThinkingText text={normalizeWhitespace(message.content)} /> - : <StructuredMessageBlock content={message.content} /> - } - </div> - )} - </div> - - {!isUser && !message.isLoading && ( - <div className="flex items-center gap-1 opacity-0 group-hover:opacity-100 transition-opacity duration-200"> - {actionIcons.map(({ icon: Icon, type, action }) => ( - <button - key={action} - onClick={() => { - const content = typeof message.content === 'string' ? message.content : (message.content as any[]).map(s => s.text || s.answer).join('\\n'); - handleAction(action, message.id, content) - }} - className="p-1.5 hover:bg-gray-700 rounded-md transition-colors text-gray-400 hover:text-gray-200" - title={type} - > - <Icon className="w-3.5 h-3.5" /> - </button> - ))} - </div> - )} - - {/* Global citations only for plain-string messages */} - {(!isUser && - !message.isLoading && - typeof message.content === 'string' && - Array.isArray((message as any).metadata?.source_documents) && - (message as any).metadata.source_documents.length > 0) && ( - <CitationsBlock docs={(message as any).metadata.source_documents} /> - )} - </div> - - {isUser && ( - <ChatBubbleAvatar - className="mt-1 flex-shrink-0 text-black" - src="https://i.pravatar.cc/40?u=user" - fallback="User" - /> - )} - </div> - </div> - ) - })} + {messages.map((message) => ( + <MessageBubble key={message.id} message={message} onAction={handleAction} /> + ))} {/* Loading indicator for new message */} {isLoading && ( diff --git a/src/components/ui/dropdown-menu.tsx b/src/components/ui/dropdown-menu.tsx deleted file mode 100644 index ec51e9cc..00000000 --- a/src/components/ui/dropdown-menu.tsx +++ /dev/null @@ -1,257 +0,0 @@ -"use client" - -import * as React from "react" -import * as DropdownMenuPrimitive from "@radix-ui/react-dropdown-menu" -import { CheckIcon, ChevronRightIcon, CircleIcon } from "lucide-react" - -import { cn } from "@/lib/utils" - -function DropdownMenu({ - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.Root>) { - return <DropdownMenuPrimitive.Root data-slot="dropdown-menu" {...props} /> -} - -function DropdownMenuPortal({ - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.Portal>) { - return ( - <DropdownMenuPrimitive.Portal data-slot="dropdown-menu-portal" {...props} /> - ) -} - -function DropdownMenuTrigger({ - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.Trigger>) { - return ( - <DropdownMenuPrimitive.Trigger - data-slot="dropdown-menu-trigger" - {...props} - /> - ) -} - -function DropdownMenuContent({ - className, - sideOffset = 4, - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.Content>) { - return ( - <DropdownMenuPrimitive.Portal> - <DropdownMenuPrimitive.Content - data-slot="dropdown-menu-content" - sideOffset={sideOffset} - className={cn( - "bg-popover text-popover-foreground data-[state=open]:animate-in data-[state=closed]:animate-out data-[state=closed]:fade-out-0 data-[state=open]:fade-in-0 data-[state=closed]:zoom-out-95 data-[state=open]:zoom-in-95 data-[side=bottom]:slide-in-from-top-2 data-[side=left]:slide-in-from-right-2 data-[side=right]:slide-in-from-left-2 data-[side=top]:slide-in-from-bottom-2 z-50 max-h-(--radix-dropdown-menu-content-available-height) min-w-[8rem] origin-(--radix-dropdown-menu-content-transform-origin) overflow-x-hidden overflow-y-auto rounded-md border p-1 shadow-md", - className - )} - {...props} - /> - </DropdownMenuPrimitive.Portal> - ) -} - -function DropdownMenuGroup({ - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.Group>) { - return ( - <DropdownMenuPrimitive.Group data-slot="dropdown-menu-group" {...props} /> - ) -} - -function DropdownMenuItem({ - className, - inset, - variant = "default", - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.Item> & { - inset?: boolean - variant?: "default" | "destructive" -}) { - return ( - <DropdownMenuPrimitive.Item - data-slot="dropdown-menu-item" - data-inset={inset} - data-variant={variant} - className={cn( - "focus:bg-accent focus:text-accent-foreground data-[variant=destructive]:text-destructive data-[variant=destructive]:focus:bg-destructive/10 dark:data-[variant=destructive]:focus:bg-destructive/20 data-[variant=destructive]:focus:text-destructive data-[variant=destructive]:*:[svg]:!text-destructive [&_svg:not([class*='text-'])]:text-muted-foreground relative flex cursor-default items-center gap-2 rounded-sm px-2 py-1.5 text-sm outline-hidden select-none data-[disabled]:pointer-events-none data-[disabled]:opacity-50 data-[inset]:pl-8 [&_svg]:pointer-events-none [&_svg]:shrink-0 [&_svg:not([class*='size-'])]:size-4", - className - )} - {...props} - /> - ) -} - -function DropdownMenuCheckboxItem({ - className, - children, - checked, - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.CheckboxItem>) { - return ( - <DropdownMenuPrimitive.CheckboxItem - data-slot="dropdown-menu-checkbox-item" - className={cn( - "focus:bg-accent focus:text-accent-foreground relative flex cursor-default items-center gap-2 rounded-sm py-1.5 pr-2 pl-8 text-sm outline-hidden select-none data-[disabled]:pointer-events-none data-[disabled]:opacity-50 [&_svg]:pointer-events-none [&_svg]:shrink-0 [&_svg:not([class*='size-'])]:size-4", - className - )} - checked={checked} - {...props} - > - <span className="pointer-events-none absolute left-2 flex size-3.5 items-center justify-center"> - <DropdownMenuPrimitive.ItemIndicator> - <CheckIcon className="size-4" /> - </DropdownMenuPrimitive.ItemIndicator> - </span> - {children} - </DropdownMenuPrimitive.CheckboxItem> - ) -} - -function DropdownMenuRadioGroup({ - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.RadioGroup>) { - return ( - <DropdownMenuPrimitive.RadioGroup - data-slot="dropdown-menu-radio-group" - {...props} - /> - ) -} - -function DropdownMenuRadioItem({ - className, - children, - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.RadioItem>) { - return ( - <DropdownMenuPrimitive.RadioItem - data-slot="dropdown-menu-radio-item" - className={cn( - "focus:bg-accent focus:text-accent-foreground relative flex cursor-default items-center gap-2 rounded-sm py-1.5 pr-2 pl-8 text-sm outline-hidden select-none data-[disabled]:pointer-events-none data-[disabled]:opacity-50 [&_svg]:pointer-events-none [&_svg]:shrink-0 [&_svg:not([class*='size-'])]:size-4", - className - )} - {...props} - > - <span className="pointer-events-none absolute left-2 flex size-3.5 items-center justify-center"> - <DropdownMenuPrimitive.ItemIndicator> - <CircleIcon className="size-2 fill-current" /> - </DropdownMenuPrimitive.ItemIndicator> - </span> - {children} - </DropdownMenuPrimitive.RadioItem> - ) -} - -function DropdownMenuLabel({ - className, - inset, - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.Label> & { - inset?: boolean -}) { - return ( - <DropdownMenuPrimitive.Label - data-slot="dropdown-menu-label" - data-inset={inset} - className={cn( - "px-2 py-1.5 text-sm font-medium data-[inset]:pl-8", - className - )} - {...props} - /> - ) -} - -function DropdownMenuSeparator({ - className, - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.Separator>) { - return ( - <DropdownMenuPrimitive.Separator - data-slot="dropdown-menu-separator" - className={cn("bg-border -mx-1 my-1 h-px", className)} - {...props} - /> - ) -} - -function DropdownMenuShortcut({ - className, - ...props -}: React.ComponentProps<"span">) { - return ( - <span - data-slot="dropdown-menu-shortcut" - className={cn( - "text-muted-foreground ml-auto text-xs tracking-widest", - className - )} - {...props} - /> - ) -} - -function DropdownMenuSub({ - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.Sub>) { - return <DropdownMenuPrimitive.Sub data-slot="dropdown-menu-sub" {...props} /> -} - -function DropdownMenuSubTrigger({ - className, - inset, - children, - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.SubTrigger> & { - inset?: boolean -}) { - return ( - <DropdownMenuPrimitive.SubTrigger - data-slot="dropdown-menu-sub-trigger" - data-inset={inset} - className={cn( - "focus:bg-accent focus:text-accent-foreground data-[state=open]:bg-accent data-[state=open]:text-accent-foreground flex cursor-default items-center rounded-sm px-2 py-1.5 text-sm outline-hidden select-none data-[inset]:pl-8", - className - )} - {...props} - > - {children} - <ChevronRightIcon className="ml-auto size-4" /> - </DropdownMenuPrimitive.SubTrigger> - ) -} - -function DropdownMenuSubContent({ - className, - ...props -}: React.ComponentProps<typeof DropdownMenuPrimitive.SubContent>) { - return ( - <DropdownMenuPrimitive.SubContent - data-slot="dropdown-menu-sub-content" - className={cn( - "bg-popover text-popover-foreground data-[state=open]:animate-in data-[state=closed]:animate-out data-[state=closed]:fade-out-0 data-[state=open]:fade-in-0 data-[state=closed]:zoom-out-95 data-[state=open]:zoom-in-95 data-[side=bottom]:slide-in-from-top-2 data-[side=left]:slide-in-from-right-2 data-[side=right]:slide-in-from-left-2 data-[side=top]:slide-in-from-bottom-2 z-50 min-w-[8rem] origin-(--radix-dropdown-menu-content-transform-origin) overflow-hidden rounded-md border p-1 shadow-lg", - className - )} - {...props} - /> - ) -} - -export { - DropdownMenu, - DropdownMenuPortal, - DropdownMenuTrigger, - DropdownMenuContent, - DropdownMenuGroup, - DropdownMenuLabel, - DropdownMenuItem, - DropdownMenuCheckboxItem, - DropdownMenuRadioGroup, - DropdownMenuRadioItem, - DropdownMenuSeparator, - DropdownMenuShortcut, - DropdownMenuSub, - DropdownMenuSubTrigger, - DropdownMenuSubContent, -} diff --git a/src/components/ui/empty-chat-state.tsx b/src/components/ui/empty-chat-state.tsx deleted file mode 100644 index 30462d96..00000000 --- a/src/components/ui/empty-chat-state.tsx +++ /dev/null @@ -1,292 +0,0 @@ -"use client"; - -import { useEffect, useRef, useCallback } from "react"; -import { useState } from "react"; -import { Textarea } from "@/components/ui/textarea"; -import { cn } from "@/lib/utils"; -import { - ArrowUpIcon, - Paperclip, - PlusIcon, - X, - FileText, -} from "lucide-react"; -import { AttachedFile } from "@/lib/types"; - -interface UseAutoResizeTextareaProps { - minHeight: number; - maxHeight?: number; -} - -function useAutoResizeTextarea({ - minHeight, - maxHeight, -}: UseAutoResizeTextareaProps) { - const textareaRef = useRef<HTMLTextAreaElement>(null); - - const adjustHeight = useCallback( - (reset?: boolean) => { - const textarea = textareaRef.current; - if (!textarea) return; - - if (reset) { - textarea.style.height = `${minHeight}px`; - return; - } - - // Temporarily shrink to get the right scrollHeight - textarea.style.height = `${minHeight}px`; - - // Calculate new height - const newHeight = Math.max( - minHeight, - Math.min( - textarea.scrollHeight, - maxHeight ?? Number.POSITIVE_INFINITY - ) - ); - - textarea.style.height = `${newHeight}px`; - }, - [minHeight, maxHeight] - ); - - useEffect(() => { - // Set initial height - const textarea = textareaRef.current; - if (textarea) { - textarea.style.height = `${minHeight}px`; - } - }, [minHeight]); - - // Adjust height on window resize - useEffect(() => { - const handleResize = () => adjustHeight(); - window.addEventListener("resize", handleResize); - return () => window.removeEventListener("resize", handleResize); - }, [adjustHeight]); - - return { textareaRef, adjustHeight }; -} - -interface EmptyChatStateProps { - onSendMessage: (message: string, attachedFiles?: AttachedFile[]) => void; - disabled?: boolean; - placeholder?: string; -} - -export function EmptyChatState({ - onSendMessage, - disabled = false, - placeholder = "Ask localgpt a question..." -}: EmptyChatStateProps) { - const [value, setValue] = useState(""); - const [attachedFiles, setAttachedFiles] = useState<AttachedFile[]>([]); - const fileInputRef = useRef<HTMLInputElement>(null); - const { textareaRef, adjustHeight } = useAutoResizeTextarea({ - minHeight: 60, - maxHeight: 200, - }); - - const handleSend = () => { - if ((value.trim() || attachedFiles.length > 0) && !disabled) { - onSendMessage(value.trim(), attachedFiles); - setValue(""); - setAttachedFiles([]); - adjustHeight(true); - } - }; - - const handleKeyDown = (e: React.KeyboardEvent<HTMLTextAreaElement>) => { - if (e.key === "Enter" && !e.shiftKey) { - e.preventDefault(); - handleSend(); - } - }; - - const handleFileAttach = () => { - fileInputRef.current?.click(); - }; - - const handleFileChange = (e: React.ChangeEvent<HTMLInputElement>) => { - const files = e.target.files; - if (!files) return; - - const newFiles: AttachedFile[] = []; - for (let i = 0; i < files.length; i++) { - const file = files[i]; - if (file.type === 'application/pdf' || - file.type === 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' || - file.type === 'application/msword' || - file.type === 'text/html' || - file.type === 'text/markdown' || - file.type === 'text/plain' || - file.name.toLowerCase().endsWith('.pdf') || - file.name.toLowerCase().endsWith('.docx') || - file.name.toLowerCase().endsWith('.doc') || - file.name.toLowerCase().endsWith('.html') || - file.name.toLowerCase().endsWith('.htm') || - file.name.toLowerCase().endsWith('.md') || - file.name.toLowerCase().endsWith('.txt')) { - newFiles.push({ - id: crypto.randomUUID(), - name: file.name, - size: file.size, - type: file.type, - file: file, - }); - } - } - - setAttachedFiles(prev => [...prev, ...newFiles]); - - // Reset the input - if (fileInputRef.current) { - fileInputRef.current.value = ''; - } - - // --- NEW: Immediately trigger upload when files are selected --- - if (newFiles.length > 0) { - onSendMessage("", newFiles); - // Clear the local attachment state as the parent now handles it - setAttachedFiles([]); - } - }; - - const removeFile = (fileId: string) => { - setAttachedFiles(prev => prev.filter(f => f.id !== fileId)); - }; - - const formatFileSize = (bytes: number) => { - if (bytes === 0) return '0 Bytes'; - const k = 1024; - const sizes = ['Bytes', 'KB', 'MB', 'GB']; - const i = Math.floor(Math.log(bytes) / Math.log(k)); - return parseFloat((bytes / Math.pow(k, i)).toFixed(2)) + ' ' + sizes[i]; - }; - - return ( - <div className="flex flex-col items-center justify-center h-full w-full max-w-4xl mx-auto p-4 space-y-8"> - <h1 className="text-4xl font-bold text-white"> - What can I help you find? - </h1> - - <div className="w-full"> - {/* Attached Files Display */} - {attachedFiles.length > 0 && ( - <div className="mb-4 space-y-2"> - <div className="text-sm text-gray-400 font-medium">Attached Files:</div> - <div className="space-y-2"> - {attachedFiles.map((file) => ( - <div key={file.id} className="flex items-center gap-3 bg-gray-800 rounded-lg p-3"> - <FileText className="w-5 h-5 text-red-400" /> - <div className="flex-1 min-w-0"> - <div className="text-sm text-white truncate">{file.name}</div> - <div className="text-xs text-gray-400">{formatFileSize(file.size)}</div> - </div> - {/* The remove button is commented out as the parent will manage the state now */} - {/* <button - onClick={() => removeFile(file.id)} - className="p-1 hover:bg-gray-700 rounded transition-colors" - > - <X className="w-4 h-4 text-gray-400 hover:text-white" /> - </button> */} - </div> - ))} - </div> - </div> - )} - - <div className="relative bg-neutral-900 rounded-xl border border-neutral-800"> - <div className="overflow-y-auto"> - <Textarea - ref={textareaRef} - value={value} - onChange={(e) => { - setValue(e.target.value); - adjustHeight(); - }} - onKeyDown={handleKeyDown} - placeholder={attachedFiles.length > 0 ? "Ask questions about your attached files..." : placeholder} - disabled={disabled} - className={cn( - "w-full px-4 py-3", - "resize-none", - "bg-transparent", - "border-none", - "text-white text-sm", - "focus:outline-none", - "focus-visible:ring-0 focus-visible:ring-offset-0", - "placeholder:text-neutral-500 placeholder:text-sm", - "min-h-[60px]", - disabled && "opacity-50 cursor-not-allowed" - )} - style={{ - overflow: "hidden", - }} - /> - </div> - - {/* Hidden file input */} - <input - ref={fileInputRef} - type="file" - accept=".pdf,.docx,.doc,.html,.htm,.md,.txt" - multiple - onChange={handleFileChange} - className="hidden" - /> - - <div className="flex items-center justify-between p-3"> - <div className="flex items-center gap-2"> - <button - type="button" - onClick={handleFileAttach} - disabled={disabled} - className="group p-2 hover:bg-neutral-800 rounded-lg transition-colors flex items-center gap-1 disabled:opacity-50 disabled:cursor-not-allowed" - title="Attach PDF files" - > - <Paperclip className="w-4 h-4 text-white" /> - <span className="text-xs text-zinc-400 hidden group-hover:inline transition-opacity"> - Attach PDF - </span> - </button> - </div> - <div className="flex items-center gap-2"> - <button - type="button" - disabled={disabled} - className="px-2 py-1 rounded-lg text-sm text-zinc-400 transition-colors border border-dashed border-zinc-700 hover:border-zinc-600 hover:bg-zinc-800 flex items-center justify-between gap-1 disabled:opacity-50 disabled:cursor-not-allowed" - > - <PlusIcon className="w-4 h-4" /> - Project - </button> - <button - type="button" - onClick={handleSend} - disabled={disabled || (!value.trim() && attachedFiles.length === 0)} - className={cn( - "px-1.5 py-1.5 rounded-lg text-sm transition-colors border border-zinc-700 hover:border-zinc-600 hover:bg-zinc-800 flex items-center justify-between gap-1", - (value.trim() || attachedFiles.length > 0) && !disabled - ? "bg-white text-black hover:bg-gray-200" - : "text-zinc-400", - "disabled:opacity-50 disabled:cursor-not-allowed" - )} - > - <ArrowUpIcon - className={cn( - "w-4 h-4", - (value.trim() || attachedFiles.length > 0) && !disabled - ? "text-black" - : "text-zinc-400" - )} - /> - <span className="sr-only">Send</span> - </button> - </div> - </div> - </div> - </div> - </div> - ); -} \ No newline at end of file diff --git a/src/components/ui/localgpt-chat.tsx b/src/components/ui/localgpt-chat.tsx deleted file mode 100644 index cb24fe64..00000000 --- a/src/components/ui/localgpt-chat.tsx +++ /dev/null @@ -1,170 +0,0 @@ -"use client"; - -import { useEffect, useRef, useCallback } from "react"; -import { useState } from "react"; -import { Textarea } from "@/components/ui/textarea"; -import { cn } from "@/lib/utils"; -import { - ArrowUpIcon, - Paperclip, - PlusIcon, -} from "lucide-react"; - -interface UseAutoResizeTextareaProps { - minHeight: number; - maxHeight?: number; -} - -function useAutoResizeTextarea({ - minHeight, - maxHeight, -}: UseAutoResizeTextareaProps) { - const textareaRef = useRef<HTMLTextAreaElement>(null); - - const adjustHeight = useCallback( - (reset?: boolean) => { - const textarea = textareaRef.current; - if (!textarea) return; - - if (reset) { - textarea.style.height = `${minHeight}px`; - return; - } - - // Temporarily shrink to get the right scrollHeight - textarea.style.height = `${minHeight}px`; - - // Calculate new height - const newHeight = Math.max( - minHeight, - Math.min( - textarea.scrollHeight, - maxHeight ?? Number.POSITIVE_INFINITY - ) - ); - - textarea.style.height = `${newHeight}px`; - }, - [minHeight, maxHeight] - ); - - useEffect(() => { - // Set initial height - const textarea = textareaRef.current; - if (textarea) { - textarea.style.height = `${minHeight}px`; - } - }, [minHeight]); - - // Adjust height on window resize - useEffect(() => { - const handleResize = () => adjustHeight(); - window.addEventListener("resize", handleResize); - return () => window.removeEventListener("resize", handleResize); - }, [adjustHeight]); - - return { textareaRef, adjustHeight }; -} - -export function LocalGPTChat() { - const [value, setValue] = useState(""); - const { textareaRef, adjustHeight } = useAutoResizeTextarea({ - minHeight: 60, - maxHeight: 200, - }); - - const handleKeyDown = (e: React.KeyboardEvent<HTMLTextAreaElement>) => { - if (e.key === "Enter" && !e.shiftKey) { - e.preventDefault(); - if (value.trim()) { - setValue(""); - adjustHeight(true); - } - } - }; - - return ( - <div className="flex flex-col items-center w-full max-w-4xl mx-auto p-4 space-y-8"> - <h1 className="text-4xl font-bold text-white"> - What can I help you find? - </h1> - - <div className="w-full"> - <div className="relative bg-neutral-900 rounded-xl border border-neutral-800"> - <div className="overflow-y-auto"> - <Textarea - ref={textareaRef} - value={value} - onChange={(e) => { - setValue(e.target.value); - adjustHeight(); - }} - onKeyDown={handleKeyDown} - placeholder="Ask localgpt a question..." - className={cn( - "w-full px-4 py-3", - "resize-none", - "bg-transparent", - "border-none", - "text-white text-sm", - "focus:outline-none", - "focus-visible:ring-0 focus-visible:ring-offset-0", - "placeholder:text-neutral-500 placeholder:text-sm", - "min-h-[60px]" - )} - style={{ - overflow: "hidden", - }} - /> - </div> - - <div className="flex items-center justify-between p-3"> - <div className="flex items-center gap-2"> - <button - type="button" - className="group p-2 hover:bg-neutral-800 rounded-lg transition-colors flex items-center gap-1" - > - <Paperclip className="w-4 h-4 text-white" /> - <span className="text-xs text-zinc-400 hidden group-hover:inline transition-opacity"> - Attach - </span> - </button> - </div> - <div className="flex items-center gap-2"> - <button - type="button" - className="px-2 py-1 rounded-lg text-sm text-zinc-400 transition-colors border border-dashed border-zinc-700 hover:border-zinc-600 hover:bg-zinc-800 flex items-center justify-between gap-1" - > - <PlusIcon className="w-4 h-4" /> - Project - </button> - <button - type="button" - className={cn( - "px-1.5 py-1.5 rounded-lg text-sm transition-colors border border-zinc-700 hover:border-zinc-600 hover:bg-zinc-800 flex items-center justify-between gap-1", - value.trim() - ? "bg-white text-black" - : "text-zinc-400" - )} - > - <ArrowUpIcon - className={cn( - "w-4 h-4", - value.trim() - ? "text-black" - : "text-zinc-400" - )} - /> - <span className="sr-only">Send</span> - </button> - </div> - </div> - </div> - - - </div> - </div> - ); -} - - \ No newline at end of file diff --git a/src/components/ui/message-loading.tsx b/src/components/ui/message-loading.tsx deleted file mode 100644 index dacc2924..00000000 --- a/src/components/ui/message-loading.tsx +++ /dev/null @@ -1,48 +0,0 @@ -"use client" - -function MessageLoading() { - return ( - <svg - width="24" - height="24" - viewBox="0 0 24 24" - xmlns="http://www.w3.org/2000/svg" - className="text-foreground" - > - <circle cx="4" cy="12" r="2" fill="currentColor"> - <animate - id="spinner_qFRN" - begin="0;spinner_OcgL.end+0.25s" - attributeName="cy" - calcMode="spline" - dur="0.6s" - values="12;6;12" - keySplines=".33,.66,.66,1;.33,0,.66,.33" - /> - </circle> - <circle cx="12" cy="12" r="2" fill="currentColor"> - <animate - begin="spinner_qFRN.begin+0.1s" - attributeName="cy" - calcMode="spline" - dur="0.6s" - values="12;6;12" - keySplines=".33,.66,.66,1;.33,0,.66,.33" - /> - </circle> - <circle cx="20" cy="12" r="2" fill="currentColor"> - <animate - id="spinner_OcgL" - begin="spinner_qFRN.begin+0.2s" - attributeName="cy" - calcMode="spline" - dur="0.6s" - values="12;6;12" - keySplines=".33,.66,.66,1;.33,0,.66,.33" - /> - </circle> - </svg> - ); -} - -export { MessageLoading }; \ No newline at end of file diff --git a/src/components/ui/quick-chat.tsx b/src/components/ui/quick-chat.tsx index c563a69a..d86bd09a 100644 --- a/src/components/ui/quick-chat.tsx +++ b/src/components/ui/quick-chat.tsx @@ -2,7 +2,7 @@ import React, { useState, useEffect } from 'react'; import { ChatInput } from '@/components/ui/chat-input'; -import { chatAPI, ChatMessage } from '@/lib/api'; +import { chatAPI, ChatMessage, DEFAULT_GENERATION_MODEL, generateUUID } from '@/lib/api'; import { ConversationPage } from '@/components/ui/conversation-page'; import { ChatSettingsModal } from '@/components/ui/chat-settings-modal'; @@ -12,15 +12,29 @@ interface QuickChatProps { className?: string; } +const QC_MODEL_KEY = 'localgpt.quickChat.model'; + export function QuickChat({ sessionId: externalSessionId, onSessionChange, className="" }: QuickChatProps) { const [messages, setMessages] = useState<ChatMessage[]>([]); const [isLoading, setIsLoading] = useState(false); - const [sessionId, setSessionId] = useState<string | undefined>(externalSessionId); + // Starts undefined even when a session id is provided at mount, so the sync + // effect below passes its guard and loads that session's messages. + const [sessionId, setSessionId] = useState<string | undefined>(undefined); const [generationModels, setGenerationModels] = useState<string[]>([]); - const [selectedModel, setSelectedModel] = useState<string>(''); + const [selectedModel, setSelectedModel] = useState<string>(() => { + if (typeof window === 'undefined') return ''; + return window.localStorage.getItem(QC_MODEL_KEY) || ''; + }); const [showSettings, setShowSettings] = useState(false); const api = chatAPI; + // Persist the chosen model so it survives unmount/mode switches + useEffect(() => { + if (selectedModel) { + try { window.localStorage.setItem(QC_MODEL_KEY, selectedModel); } catch {} + } + }, [selectedModel]); + // ๐Ÿ”„ Sync prop -> state: when sidebar selects a different session, update local session and reset chat window useEffect(() => { if (externalSessionId && externalSessionId !== sessionId) { @@ -48,8 +62,9 @@ export function QuickChat({ sessionId: externalSessionId, onSessionChange, class const resp = await api.getModels(); setGenerationModels(resp.generation_models||[]); if(resp.generation_models && resp.generation_models.length>0){ - const def = resp.generation_models.find((m:string)=>m==='qwen3:8b'); - setSelectedModel(def || resp.generation_models[0]); + const def = resp.generation_models.find((m:string)=>m===DEFAULT_GENERATION_MODEL); + // Keep a persisted selection only if the backend still offers it + setSelectedModel(prev => (prev && resp.generation_models.includes(prev)) ? prev : (def || resp.generation_models[0])); } }catch(e){console.warn('Failed to load models',e);} })(); @@ -59,7 +74,7 @@ export function QuickChat({ sessionId: externalSessionId, onSessionChange, class if (!content.trim()) return; const userMsg: ChatMessage = { - id: crypto.randomUUID(), + id: generateUUID(), content, sender: 'user', timestamp: new Date().toISOString(), @@ -75,8 +90,8 @@ export function QuickChat({ sessionId: externalSessionId, onSessionChange, class const newSess = await api.createSession('Quick Chat'); activeSessionId = newSess.id; setSessionId(activeSessionId); - if(onSessionChange){ - onSessionChange(newSess); + if(onSessionChange){ + onSessionChange(newSess); } } catch (err) { console.error('Failed to create quick-chat session', err); @@ -85,25 +100,69 @@ export function QuickChat({ sessionId: externalSessionId, onSessionChange, class try { const history = api.messagesToHistory(messages); - const resp = await api.sendMessage({ message: content, conversation_history: history, model: selectedModel }); - const assistantMsg: ChatMessage = { - id: crypto.randomUUID(), - content: resp.response, - sender: 'assistant', - timestamp: new Date().toISOString(), - }; - setMessages((prev) => [...prev, assistantMsg]); + // Stream token-by-token. The placeholder is appended first and each + // token event replaces it with a NEW message object โ€” the memoized + // bubbles re-render only when their message identity changes. + const assistantId = generateUUID(); + setMessages((prev) => [...prev, { + id: assistantId, + content: '', + sender: 'assistant', + timestamp: new Date().toISOString(), + }]); + + let finalText = ''; + let streamError: string | null = null; + await api.streamChatMessage( + { message: content, conversation_history: history, model: selectedModel }, + (evt) => { + if (evt.type === 'token') { + const tok: string = evt.data?.text || ''; + if (!tok) return; + finalText += tok; + const snapshot = finalText; + setMessages((prev) => prev.map((m) => + m.id === assistantId ? { ...m, content: snapshot } : m)); + } else if (evt.type === 'complete') { + const answer = typeof evt.data?.response === 'string' && evt.data.response + ? evt.data.response : finalText; + finalText = answer; + setMessages((prev) => prev.map((m) => + m.id === assistantId ? { ...m, content: answer } : m)); + } else if (evt.type === 'error') { + streamError = typeof evt.data?.error === 'string' ? evt.data.error : 'Quick chat failed'; + } + }, + ); + if (streamError) throw new Error(streamError); + const resp = { response: finalText }; + + // /chat itself persists nothing โ€” save the turn against the session so it + // survives reloads and shows up in the sidebar. Skip gracefully if the + // session could not be created above. + if (activeSessionId) { + try { + const saved = await api.saveStreamedTurn(activeSessionId, content, resp.response); + if (onSessionChange && saved?.session) { + onSessionChange(saved.session); + } + } catch (saveErr) { + console.error('Failed to persist quick-chat turn', saveErr); + } + } } catch (err) { console.error('Quick chat failed', err); + const errText = err instanceof Error ? err.message : 'Quick chat failed'; + setMessages((prev) => [...prev, { + id: generateUUID(), + content: `โŒ ${errText}`, + sender: 'assistant', + timestamp: new Date().toISOString(), + }]); } finally { setIsLoading(false); } - - // if session existed externally and callback provided, still sync id - if(onSessionChange && activeSessionId && activeSessionId!==externalSessionId){ - // no additional action; already sent on creation - } }; const showEmptyState = messages.length === 0 && !isLoading diff --git a/src/components/ui/separator.tsx b/src/components/ui/separator.tsx deleted file mode 100644 index 275381ca..00000000 --- a/src/components/ui/separator.tsx +++ /dev/null @@ -1,28 +0,0 @@ -"use client" - -import * as React from "react" -import * as SeparatorPrimitive from "@radix-ui/react-separator" - -import { cn } from "@/lib/utils" - -function Separator({ - className, - orientation = "horizontal", - decorative = true, - ...props -}: React.ComponentProps<typeof SeparatorPrimitive.Root>) { - return ( - <SeparatorPrimitive.Root - data-slot="separator" - decorative={decorative} - orientation={orientation} - className={cn( - "bg-border shrink-0 data-[orientation=horizontal]:h-px data-[orientation=horizontal]:w-full data-[orientation=vertical]:h-full data-[orientation=vertical]:w-px", - className - )} - {...props} - /> - ) -} - -export { Separator } diff --git a/src/components/ui/session-chat.tsx b/src/components/ui/session-chat.tsx index c14aa1e7..810f33fc 100644 --- a/src/components/ui/session-chat.tsx +++ b/src/components/ui/session-chat.tsx @@ -1,12 +1,10 @@ "use client" -import * as React from "react" import { ConversationPage } from "./conversation-page" import { ChatInput } from "./chat-input" -import { EmptyChatState } from "./empty-chat-state" -import { ChatMessage, ChatSession, chatAPI, generateUUID } from "@/lib/api" +import { ChatMessage, ChatSession, chatAPI, generateUUID, DEFAULT_GENERATION_MODEL } from "@/lib/api" import { AttachedFile } from "@/lib/types" -import { useEffect, useState, forwardRef, useImperativeHandle, useCallback } from "react" +import { useEffect, useState, useRef, useMemo, useCallback } from "react" import { normalizeStreamingToken } from "@/utils/textNormalization" import { Button } from "./button" import type { Step } from '@/lib/api' @@ -18,49 +16,74 @@ import { Database } from 'lucide-react' interface SessionChatProps { sessionId?: string onSessionChange?: (session: ChatSession) => void - onNewMessage?: (message: ChatMessage) => void className?: string } -// Export sendMessage function for parent components -export interface SessionChatRef { - sendMessage: (content: string, attachedFiles?: AttachedFile[]) => Promise<void> - currentSession: ChatSession | null -} - // Helper to shorten long titles const truncate = (str: string, n: number = 18) => str.length > n ? str.slice(0, n) + 'โ€ฆ' : str; -export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ +// Chat settings persist in localStorage so they survive unmount/mode switches +const CHAT_SETTINGS_KEY = 'localgpt.chatSettings' + +interface PersistedChatSettings { + composeSubAnswers?: boolean + enableDecompose?: boolean + enableAiRerank?: boolean + enableContextExpand?: boolean + enableStream?: boolean + enableVerify?: boolean + forceDocs?: boolean + provencePrune?: boolean + retrievalK?: number + contextWindowSize?: number + rerankerTopK?: number + searchType?: string + selectedModel?: string +} + +const loadPersistedChatSettings = (): PersistedChatSettings => { + if (typeof window === 'undefined') return {} + try { + return JSON.parse(window.localStorage.getItem(CHAT_SETTINGS_KEY) || '{}') + } catch { + return {} + } +} + +export function SessionChat({ sessionId, onSessionChange, - onNewMessage, className = "" -}, ref) => { +}: SessionChatProps) { const [messages, setMessages] = useState<ChatMessage[]>([]) const [isLoading, setIsLoading] = useState(false) const [currentSession, setCurrentSession] = useState<ChatSession | null>(null) const [error, setError] = useState<string | null>(null) const [uploadedFiles, setUploadedFiles] = useState<{filename: string, stored_path: string}[]>([]) const [isIndexed, setIsIndexed] = useState(false) - const [composeSubAnswers, setComposeSubAnswers] = useState<boolean>(true) - const [enableDecompose, setEnableDecompose] = useState<boolean>(true) - const [enableAiRerank, setEnableAiRerank] = useState<boolean>(true) - const [enableContextExpand, setEnableContextExpand] = useState<boolean>(true) - const [enableStream, setEnableStream] = useState<boolean>(true) - const [enableVerify, setEnableVerify] = useState<boolean>(true) + // Settings load once from localStorage; a save effect further down writes them back on change. + const persistedSettings = useMemo(loadPersistedChatSettings, []) + // Off, matching PIPELINE_CONFIGS["default"].query_decomposition (arm H: + // pooled_first_stage with compose_from_sub_answers False) โ€” see eval/DECISIONS.md. + const [composeSubAnswers, setComposeSubAnswers] = useState<boolean>(() => persistedSettings.composeSubAnswers ?? false) + const [enableDecompose, setEnableDecompose] = useState<boolean>(() => persistedSettings.enableDecompose ?? true) + // On by default, matching PIPELINE_CONFIGS["default"].reranker.enabled (arm G) + // โ€” see eval/DECISIONS.md. Loads Qwen3-Reranker-4B lazily. + const [enableAiRerank, setEnableAiRerank] = useState<boolean>(() => persistedSettings.enableAiRerank ?? true) + const [enableContextExpand, setEnableContextExpand] = useState<boolean>(() => persistedSettings.enableContextExpand ?? true) + const [enableStream, setEnableStream] = useState<boolean>(() => persistedSettings.enableStream ?? true) + const [enableVerify, setEnableVerify] = useState<boolean>(() => persistedSettings.enableVerify ?? true) // Force RAG toggle - const [forceDocs, setForceDocs] = useState<boolean>(false) + const [forceDocs, setForceDocs] = useState<boolean>(() => persistedSettings.forceDocs ?? false) // Provence pruning toggle - const [provencePrune, setProvencePrune] = useState<boolean>(false) - - // โœจ NEW RETRIEVAL PARAMETERS - const [retrievalK, setRetrievalK] = useState<number>(20) - const [contextWindowSize, setContextWindowSize] = useState<number>(1) - const [rerankerTopK, setRerankerTopK] = useState<number>(10) - const [searchType, setSearchType] = useState<string>('hybrid') + const [provencePrune, setProvencePrune] = useState<boolean>(() => persistedSettings.provencePrune ?? false) + + const [retrievalK, setRetrievalK] = useState<number>(() => persistedSettings.retrievalK ?? 20) + const [contextWindowSize, setContextWindowSize] = useState<number>(() => persistedSettings.contextWindowSize ?? 1) + const [rerankerTopK, setRerankerTopK] = useState<number>(() => persistedSettings.rerankerTopK ?? 10) + const [searchType, setSearchType] = useState<string>(() => persistedSettings.searchType ?? 'hybrid') const [generationModels,setGenerationModels]=useState<string[]>([]) - const [selectedModel,setSelectedModel]=useState<string>('qwen3:8b') + const [selectedModel,setSelectedModel]=useState<string>(() => persistedSettings.selectedModel ?? DEFAULT_GENERATION_MODEL) const [currentIndexId, setCurrentIndexId] = useState<string | null>(null) const [currentIndexName, setCurrentIndexName] = useState<string | null>(null) const [showSettings, setShowSettings] = useState(false) @@ -69,23 +92,49 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ const apiService = chatAPI + // Persist chat settings on change (loaded on mount above) + useEffect(() => { + try { + window.localStorage.setItem(CHAT_SETTINGS_KEY, JSON.stringify({ + composeSubAnswers, enableDecompose, enableAiRerank, enableContextExpand, + enableStream, enableVerify, forceDocs, provencePrune, + retrievalK, contextWindowSize, rerankerTopK, searchType, selectedModel, + })) + } catch {} + }, [composeSubAnswers, enableDecompose, enableAiRerank, enableContextExpand, enableStream, enableVerify, forceDocs, provencePrune, retrievalK, contextWindowSize, rerankerTopK, searchType, selectedModel]) + + // Monotonic load counter: a slow getSession for a session the user has + // already navigated away from must not clobber the one they switched to. + const loadSessionSeq = useRef(0) + // Define loadSession with useCallback before useEffect const loadSession = useCallback(async (id: string) => { + const seq = ++loadSessionSeq.current try { setError(null) + // Clear per-session state up front โ€” index/upload state from the + // previous session must never leak into this one (a stale index id + // would override the backend's session-derived table_name). + setCurrentIndexId(null) + setCurrentIndexName(null) + setUploadedFiles([]) + setIsIndexed(false) const { session, messages: sessionMessages } = await apiService.getSession(id) - + if (seq !== loadSessionSeq.current) return // superseded by a newer load + const convertedMessages = sessionMessages.map((msg: unknown) => apiService.convertDbMessage(msg as Record<string, unknown>)) setMessages(convertedMessages) setCurrentSession(session) - + if (onSessionChange) { onSessionChange(session) } - // Fetch linked indexes to know table name for streaming + // Fetch linked indexes to know table name for streaming. When the + // session has none, the up-front clear above leaves index state empty. try { const idxResp = await apiService.getSessionIndexes(id) + if (seq !== loadSessionSeq.current) return if (idxResp.indexes && idxResp.indexes.length > 0) { const lastIdxObj = idxResp.indexes[idxResp.indexes.length - 1] as any const idxId = (lastIdxObj.index_id ?? lastIdxObj.id) as string @@ -94,6 +143,7 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ } } catch {} } catch (error) { + if (seq !== loadSessionSeq.current) return console.error('Failed to load session:', error) setError('Failed to load session') } @@ -108,9 +158,14 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ loadSession(sessionId) } } else { - // Clear messages if no session + // No session: invalidate any in-flight load and clear per-session state + loadSessionSeq.current++ setMessages([]) setCurrentSession(null) + setCurrentIndexId(null) + setCurrentIndexName(null) + setUploadedFiles([]) + setIsIndexed(false) } }, [sessionId, currentSession, loadSession]) // Added missing dependencies @@ -121,14 +176,15 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ const resp=await apiService.getModels(); setGenerationModels(resp.generation_models||[]) if(resp.generation_models&&resp.generation_models.length>0){ - const def = resp.generation_models.find((m:string)=>m==='qwen3:8b'); - setSelectedModel(def || resp.generation_models[0]) + const def = resp.generation_models.find((m:string)=>m===DEFAULT_GENERATION_MODEL); + // Keep a persisted selection only if the backend still offers it + setSelectedModel(prev => (prev && resp.generation_models.includes(prev)) ? prev : (def || resp.generation_models[0])) } }catch(e){console.warn('Failed to load models',e)} })() },[apiService]) - const sendMessage = async (content: string, attachedFiles?: AttachedFile[]) => { + const sendMessage = async (content: string, attachedFiles?: AttachedFile[], opts?: { skipUserTurn?: boolean }) => { // --- Guard Clauses --- // If files are being indexed, do nothing. if (uploadedFiles.length > 0 && !isIndexed) { @@ -188,9 +244,12 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ // B) CHAT ACTION: If no files, it's a standard chat message. if (!content.trim()) return; - const userMessage = apiService.createMessage(content, 'user') - setMessages(prev => [...prev, userMessage]) - if (onNewMessage) onNewMessage(userMessage) + // Regenerate passes skipUserTurn: the original user bubble stays in place, + // so don't append (or later persist) a duplicate of it. + if (!opts?.skipUserTurn) { + const userMessage = apiService.createMessage(content, 'user') + setMessages(prev => [...prev, userMessage]) + } setIsLoading(true) @@ -214,6 +273,9 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ { key: 'analyze', label: 'Analyzing user question', status: 'pending' as const, details: '' }, { key: 'decompose', label: 'Generating sub-queries', status: 'pending' as const, details: '' }, { key: 'retrieval', label: 'Retrieving context', status: 'pending' as const, details: '' }, + // Only ever becomes visible when the evidence-sufficiency retry fires + // (retrieval.retry, roadmap 2.1) โ€” on most queries it stays pending. + { key: 'retry', label: 'Rechecking weak evidence', status: 'pending' as const, details: '' }, { key: 'rerank', label: 'Reranking results', status: 'pending' as const, details: '' }, { key: 'expand', label: 'Expanding context window', status: 'pending' as const, details: '' }, { key: 'answer', label: 'Answering sub-queries', status: 'pending' as const, details: [] }, @@ -234,6 +296,11 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ }) // keep global isLoading true so input disabled until completion + // Written from the setMessages updater when the 'complete' event lands, + // read by the deferred persistence call above. Assignment is idempotent, + // so StrictMode's double updater invocation is harmless. + const finalStepsSnapshot: { steps: Step[] | null } = { steps: null } + await apiService.streamSessionMessage( { query: content, @@ -245,7 +312,6 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ contextExpand: enableContextExpand, verify: enableVerify, model: selectedModel, - // โœจ NEW RETRIEVAL PARAMETERS retrievalK, contextWindowSize, rerankerTopK, @@ -254,7 +320,42 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ provencePrune, }, (evt) => { - console.log('STREAM EVENT:', evt.type, evt.data); // Debug log for SSE events + // ๐Ÿ’พ PERSIST + REFRESH on completion. This must happen HERE, in the + // event callback (fires exactly once per SSE event) โ€” never inside the + // setMessages updater below, which React StrictMode invokes twice and + // would persist every turn twice. The save itself is deferred one tick + // so the updater has applied and finalStepsSnapshot holds the finished + // pipeline cascade (writing a ref-like local from the updater is + // idempotent, so StrictMode double-invocation is harmless). + if (evt.type === 'complete' && activeSessionId && !opts?.skipUserTurn) { + const answerText = typeof evt.data.answer === 'string' ? evt.data.answer : ''; + const sourceDocs = Array.isArray(evt.data.source_documents) ? evt.data.source_documents : []; + setTimeout(async () => { + try { + if (answerText) { + await apiService.saveStreamedTurn( + activeSessionId as string, content, answerText, sourceDocs, + finalStepsSnapshot.steps ?? undefined, + ); + } + const { session } = await apiService.getSession(activeSessionId as string); + setCurrentSession(session); + if (onSessionChange) { + onSessionChange(session); + } + } catch (error) { + console.error('Failed to persist/refresh session after completion:', error); + } + }, 0); + } + // The backend emits a terminal 'error' event and then closes the + // stream normally โ€” surface it here; the updater below fails the + // placeholder's unfinished steps so the cascade doesn't hang. + if (evt.type === 'error') { + const errText = typeof evt.data?.error === 'string' ? evt.data.error : 'Something went wrong while generating the answer.'; + setError(errText); + setIsLoading(false); + } setMessages(prev => prev.map(m => { if (m.id !== placeholder.id) return m; const steps = [...(m.content as any).steps]; @@ -288,6 +389,33 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ } return { ...m, content: { steps } }; } + if (evt.type === 'retrieval_retry') { + const tidx = steps.findIndex(s => s.key === 'retry'); + if (tidx !== -1) { + steps[tidx].status = 'done'; + steps[tidx].details = evt.data?.kept === 'retry' + ? `Weak evidence (${evt.data?.signal} ${evt.data?.score_before}); retried and kept the better result set.` + : `Weak evidence (${evt.data?.signal} ${evt.data?.score_before}); retry did not improve it, kept the original.`; + } + return { ...m, content: { steps } }; + } + if (evt.type === 'document_escalation') { + // Roadmap 4.1. Reported on the existing "weak evidence" step + // rather than a new one: escalation only ever happens because + // the evidence was still weak after the retry, and adding an + // array entry would shift the positional step indices the rest + // of this handler relies on. + const tidx = steps.findIndex(s => s.key === 'retry'); + if (tidx !== -1) { + const prior = typeof steps[tidx].details === 'string' ? steps[tidx].details : ''; + const note = `Escalated to the full document "${evt.data?.document_name}" ` + + `(${evt.data?.chunks_used}/${evt.data?.chunks_total} chunks, ` + + `~${evt.data?.approx_tokens} tokens${evt.data?.truncated ? ', truncated' : ''}).`; + steps[tidx].status = 'done'; + steps[tidx].details = prior ? `${prior} ${note}` : note; + } + return { ...m, content: { steps } }; + } if (evt.type === 'rerank_started') { const rrxIdx = steps.findIndex(s => s.key === 'rerank'); if (rrxIdx !== -1) { @@ -344,7 +472,6 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ steps[5].status = 'done'; steps[6].status = 'active'; steps[6].details = 'Synthesizing final answer...'; - if (isLoading) setIsLoading(false); return { ...m, content: { steps } }; } if (evt.type === 'token') { @@ -370,18 +497,12 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ } let updated = current.endsWith(tok) ? current : current + tok; updated = normalizeStreamingToken('', updated); - if (steps[finalIdx].key === 'direct') { - steps[0].details = updated; - } else { - steps[7].details = { answer: updated, source_documents: [] }; - } steps[finalIdx].details = updated; // Mark "Putting everything together" step as done once tokens start const synthIdx = steps.findIndex(s => s.key === 'synthesize'); if (synthIdx !== -1 && steps[synthIdx].status !== 'done') { steps[synthIdx].status = 'done'; } - if (isLoading) setIsLoading(false); return { ...m, content: { steps } }; } if (evt.type === 'sub_query_token') { @@ -400,7 +521,6 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ detailsArr[idx].answer = updatedAnswer; } steps[5].details = detailsArr; - if (isLoading) setIsLoading(false); return { ...m, content: { steps } }; } if (evt.type === 'complete') { @@ -414,7 +534,12 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ } else { steps[finalIdx].details = { answer: evt.data.answer, - source_documents: evt.data.source_documents || [] + source_documents: evt.data.source_documents || [], + // Roadmap 4.5: per-query token counts, carried through the + // steps snapshot into the persisted turn. Nothing renders it + // yet โ€” displaying it is tracked as backlog in + // eval/decisions/phase4-escalation-tokens.md. + token_usage: evt.data.token_usage ?? null, }; } @@ -423,25 +548,18 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ steps.forEach(s => { if (s.status !== 'done') s.status = 'done'; }); - - // ๐Ÿ”„ REFRESH SESSION: After completion, refresh session data to get updated title - if (activeSessionId) { - // Always refresh session data so updated title & message count are reflected in the UI - setTimeout(async () => { - try { - const { session } = await apiService.getSession(activeSessionId as string); - setCurrentSession(session); - if (onSessionChange) { - onSessionChange(session); - } - } catch (error) { - console.error('Failed to refresh session after completion:', error); - } - }, 100); // Small delay to ensure backend has processed the title update - } - + + finalStepsSnapshot.steps = steps; return { ...m, content: { steps }, metadata: { message_type: 'complete' } }; } + if (evt.type === 'error') { + // Terminal backend error: fail every unfinished step so the + // placeholder renders the failure instead of spinning forever. + steps.forEach(s => { + if (s.status !== 'done') s.status = 'error'; + }); + return { ...m, content: { steps }, metadata: { message_type: 'error' } }; + } if (evt.type === 'direct_answer') { const stepsDir: Step[] = [ { key: 'direct', label: 'Answering directly', status: 'active' as const, details: '' } @@ -460,7 +578,6 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ contextExpand: enableContextExpand, verify: enableVerify, model: selectedModel, - // โœจ NEW RETRIEVAL PARAMETERS retrievalK, contextWindowSize, rerankerTopK, @@ -468,30 +585,39 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ forceRag: forceDocs, provencePrune, }) - + const aiMessage: ChatMessage = { - id: response.ai_message_id || generateUUID(), + id: generateUUID(), content: response.response, sender: 'assistant', timestamp: new Date().toISOString(), - metadata: { + metadata: { message_type: 'sub_answer', - source_documents: (response as any).source_documents || [] + source_documents: response.source_documents || [] } } setMessages(prev => [...prev, aiMessage]) - - if ((response as any).session) { - const sess = (response as any).session as ChatSession - setCurrentSession(sess) - if (onSessionChange) onSessionChange(sess) + + if (response.session) { + setCurrentSession(response.session) + if (onSessionChange) onSessionChange(response.session) } - if (onNewMessage) onNewMessage(aiMessage) } } catch (error) { console.error('Failed to send message:', error) - setError('Failed to send message') + setError(error instanceof Error ? error.message : 'Failed to send message') + // A mid-stream network failure leaves the step placeholder mid-cascade โ€” + // fail its unfinished steps instead of letting it spin forever. + setMessages(prev => prev.map(m => { + if (m.metadata?.message_type !== 'in_progress') return m + const steps = (m.content as { steps: Step[] }).steps + return { + ...m, + content: { steps: steps.map(s => s.status === 'done' ? s : { ...s, status: 'error' as const }) }, + metadata: { message_type: 'error' } + } + })) } finally { setIsLoading(false) } @@ -526,15 +652,9 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ } } - // Expose functions to parent component - useImperativeHandle(ref, () => ({ - sendMessage, - currentSession - })) - const handleAction = async (action: string, messageId: string, messageContent: string | Record<string, any>[] | { steps: Step[] }) => { console.log(`Action ${action} on message ${messageId}`) - + switch (action) { case 'copy': await navigator.clipboard.writeText(typeof messageContent === 'string' ? messageContent : JSON.stringify(messageContent, null, 2)) @@ -544,10 +664,12 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ const messageIndex = messages.findIndex(m => m.id === messageId) if (messageIndex > 0 && messages[messageIndex].sender === 'assistant') { const userMessage = messages[messageIndex - 1] - if (userMessage.sender === 'user') { - // Remove the AI message and resend the user message + if (userMessage.sender === 'user' && typeof userMessage.content === 'string') { + // Remove the stale AI message and re-ask WITHOUT re-appending or + // re-persisting the user turn (skipUserTurn) โ€” otherwise the + // question shows up twice in the UI and in the database. setMessages(prev => prev.filter(m => m.id !== messageId)) - await sendMessage(userMessage.content as string) + await sendMessage(userMessage.content, undefined, { skipUserTurn: true }) } } break @@ -646,7 +768,7 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ {type: 'dropdown', label:'Search type', value: searchType, setter: setSearchType, options: [ {value: 'hybrid', label: 'Hybrid (Vector + FTS)'}, {value: 'vector_only', label: 'Vector Only'}, - {value: 'bm25_only', label: 'FTS Only'} + {value: 'fts_only', label: 'FTS Only'} ]}, {type: 'slider', label:'Retrieval chunks', value: retrievalK, setter: setRetrievalK, min: 5, max: 50, unit: ' chunks'}, @@ -678,6 +800,4 @@ export const SessionChat = forwardRef<SessionChatRef, SessionChatProps>(({ )} </div> ) -}) - -SessionChat.displayName = "SessionChat" \ No newline at end of file +} \ No newline at end of file diff --git a/src/components/ui/sidebar.tsx b/src/components/ui/sidebar.tsx deleted file mode 100644 index 3b481bca..00000000 --- a/src/components/ui/sidebar.tsx +++ /dev/null @@ -1,260 +0,0 @@ -"use client"; - -import { cn } from "@/lib/utils"; -import { ScrollArea } from "@/components/ui/scroll-area"; -import { motion } from "framer-motion"; -import { - ChevronsUpDown, - LogOut, - MessagesSquare, - Plus, - Settings, - UserCircle, -} from "lucide-react"; -import { Avatar, AvatarFallback } from "@/components/ui/avatar" -import { useState } from "react"; -import { Button } from "@/components/ui/button"; -import { - DropdownMenu, - DropdownMenuContent, - DropdownMenuItem, - DropdownMenuSeparator, - DropdownMenuTrigger, -} from "@/components/ui/dropdown-menu"; -import { Separator } from "@/components/ui/separator"; - -const sidebarVariants = { - open: { - width: "15rem", - }, - closed: { - width: "3.05rem", - }, -}; - -const contentVariants = { - open: { display: "block", opacity: 1 }, - closed: { display: "block", opacity: 1 }, -}; - -const variants = { - open: { - x: 0, - opacity: 1, - transition: { - x: { stiffness: 1000, velocity: -100 }, - }, - }, - closed: { - x: -20, - opacity: 0, - transition: { - x: { stiffness: 100 }, - }, - }, -}; - -const transitionProps = { - type: "tween", - ease: "easeOut", - duration: 0.2, - staggerChildren: 0.1, -}; - -const staggerVariants = { - open: { - transition: { staggerChildren: 0.03, delayChildren: 0.02 }, - }, -}; - -// Mock chat sessions data -const chatSessions = [ - { id: 1, title: "React Component Help", lastMessage: "How to create a sidebar?", timestamp: "2 min ago", isActive: true }, - { id: 2, title: "TypeScript Questions", lastMessage: "Interface vs Type", timestamp: "1 hour ago", isActive: false }, - { id: 3, title: "Next.js Setup", lastMessage: "Setting up shadcn/ui", timestamp: "3 hours ago", isActive: false }, - { id: 4, title: "Tailwind CSS", lastMessage: "Dark mode implementation", timestamp: "1 day ago", isActive: false }, - { id: 5, title: "Database Design", lastMessage: "Schema optimization", timestamp: "2 days ago", isActive: false }, -]; - -export function SessionNavBar() { - const [isCollapsed, setIsCollapsed] = useState(true); - - return ( - <motion.div - className={cn( - "sidebar fixed left-0 z-40 h-full shrink-0 border-r border-neutral-800", - )} - initial={isCollapsed ? "closed" : "open"} - animate={isCollapsed ? "closed" : "open"} - variants={sidebarVariants} - transition={transitionProps} - onMouseEnter={() => setIsCollapsed(false)} - onMouseLeave={() => setIsCollapsed(true)} - > - <motion.div - className={`relative z-40 flex text-muted-foreground h-full shrink-0 flex-col bg-black transition-all`} - variants={contentVariants} - > - <motion.ul variants={staggerVariants} className="flex h-full flex-col"> - <div className="flex grow flex-col items-center"> - {/* Header */} - <div className="flex h-[54px] w-full shrink-0 border-b border-neutral-800 p-2"> - <div className="mt-[1.5px] flex w-full"> - <DropdownMenu modal={false}> - <DropdownMenuTrigger className="w-full" asChild> - <Button - variant="ghost" - size="sm" - className="flex w-fit items-center gap-2 px-2 text-white hover:bg-neutral-800" - > - <Avatar className='rounded size-4'> - <AvatarFallback className="bg-blue-600 text-white">L</AvatarFallback> - </Avatar> - <motion.li - variants={variants} - className="flex w-fit items-center gap-2" - > - {!isCollapsed && ( - <> - <p className="text-sm font-medium text-white"> - localGPT - </p> - <ChevronsUpDown className="h-4 w-4 text-neutral-400" /> - </> - )} - </motion.li> - </Button> - </DropdownMenuTrigger> - <DropdownMenuContent align="start" className="bg-neutral-900 border-neutral-800"> - <DropdownMenuItem className="flex items-center gap-2 text-white hover:bg-neutral-800"> - <Settings className="h-4 w-4" /> Preferences - </DropdownMenuItem> - <DropdownMenuItem className="flex items-center gap-2 text-white hover:bg-neutral-800"> - <Plus className="h-4 w-4" /> New Chat - </DropdownMenuItem> - </DropdownMenuContent> - </DropdownMenu> - </div> - </div> - - {/* Chat Sessions */} - <div className="flex h-full w-full flex-col"> - <div className="flex grow flex-col gap-4"> - <ScrollArea className="h-16 grow p-2"> - <div className={cn("flex w-full flex-col gap-1")}> - {/* New Chat Button */} - <Button - variant="ghost" - className="flex h-8 w-full flex-row items-center justify-start rounded-md px-2 py-1.5 text-white hover:bg-neutral-800 mb-2" - > - <Plus className="h-4 w-4" /> - <motion.span variants={variants} className="ml-2"> - {!isCollapsed && ( - <p className="text-sm font-medium">New Chat</p> - )} - </motion.span> - </Button> - - <Separator className="w-full bg-neutral-800" /> - - {/* Chat Sessions List */} - {chatSessions.map((session) => ( - <div - key={session.id} - className={cn( - "flex h-auto w-full flex-col rounded-md px-2 py-2 transition hover:bg-neutral-800 cursor-pointer", - session.isActive && "bg-neutral-800" - )} - > - <div className="flex items-center gap-2"> - <MessagesSquare className="h-4 w-4 text-neutral-400 shrink-0" /> - <motion.div variants={variants} className="flex-1 min-w-0"> - {!isCollapsed && ( - <div className="flex flex-col gap-1"> - <p className="text-sm font-medium text-white truncate"> - {session.title} - </p> - <p className="text-xs text-neutral-400 truncate"> - {session.lastMessage} - </p> - <p className="text-xs text-neutral-500"> - {session.timestamp} - </p> - </div> - )} - </motion.div> - </div> - </div> - ))} - </div> - </ScrollArea> - </div> - - {/* Footer */} - <div className="flex flex-col p-2 border-t border-neutral-800"> - <Button - variant="ghost" - className="mt-auto flex h-8 w-full flex-row items-center rounded-md px-2 py-1.5 text-white hover:bg-neutral-800" - > - <Settings className="h-4 w-4 shrink-0" /> - <motion.span variants={variants}> - {!isCollapsed && ( - <p className="ml-2 text-sm font-medium">Settings</p> - )} - </motion.span> - </Button> - - <DropdownMenu modal={false}> - <DropdownMenuTrigger className="w-full"> - <div className="flex h-8 w-full flex-row items-center gap-2 rounded-md px-2 py-1.5 transition hover:bg-neutral-800"> - <Avatar className="size-4"> - <AvatarFallback className="bg-blue-600 text-white text-xs"> - U - </AvatarFallback> - </Avatar> - <motion.div - variants={variants} - className="flex w-full items-center gap-2" - > - {!isCollapsed && ( - <> - <p className="text-sm font-medium text-white">User</p> - <ChevronsUpDown className="ml-auto h-4 w-4 text-neutral-400" /> - </> - )} - </motion.div> - </div> - </DropdownMenuTrigger> - <DropdownMenuContent sideOffset={5} className="bg-neutral-900 border-neutral-800"> - <div className="flex flex-row items-center gap-2 p-2"> - <Avatar className="size-6"> - <AvatarFallback className="bg-blue-600 text-white"> - U - </AvatarFallback> - </Avatar> - <div className="flex flex-col text-left"> - <span className="text-sm font-medium text-white"> - User - </span> - <span className="line-clamp-1 text-xs text-neutral-400"> - user@example.com - </span> - </div> - </div> - <DropdownMenuSeparator className="bg-neutral-800" /> - <DropdownMenuItem className="flex items-center gap-2 text-white hover:bg-neutral-800"> - <UserCircle className="h-4 w-4" /> Profile - </DropdownMenuItem> - <DropdownMenuItem className="flex items-center gap-2 text-white hover:bg-neutral-800"> - <LogOut className="h-4 w-4" /> Sign out - </DropdownMenuItem> - </DropdownMenuContent> - </DropdownMenu> - </div> - </div> - </div> - </motion.ul> - </motion.div> - </motion.div> - ); -} \ No newline at end of file diff --git a/src/components/ui/skeleton.tsx b/src/components/ui/skeleton.tsx deleted file mode 100644 index 32ea0ef7..00000000 --- a/src/components/ui/skeleton.tsx +++ /dev/null @@ -1,13 +0,0 @@ -import { cn } from "@/lib/utils" - -function Skeleton({ className, ...props }: React.ComponentProps<"div">) { - return ( - <div - data-slot="skeleton" - className={cn("bg-accent animate-pulse rounded-md", className)} - {...props} - /> - ) -} - -export { Skeleton } diff --git a/src/components/ui/textarea.tsx b/src/components/ui/textarea.tsx deleted file mode 100644 index 7f21b5e7..00000000 --- a/src/components/ui/textarea.tsx +++ /dev/null @@ -1,18 +0,0 @@ -import * as React from "react" - -import { cn } from "@/lib/utils" - -function Textarea({ className, ...props }: React.ComponentProps<"textarea">) { - return ( - <textarea - data-slot="textarea" - className={cn( - "border-input placeholder:text-muted-foreground focus-visible:border-ring focus-visible:ring-ring/50 aria-invalid:ring-destructive/20 dark:aria-invalid:ring-destructive/40 aria-invalid:border-destructive dark:bg-input/30 flex field-sizing-content min-h-16 w-full rounded-md border bg-transparent px-3 py-2 text-base shadow-xs transition-[color,box-shadow] outline-none focus-visible:ring-[3px] disabled:cursor-not-allowed disabled:opacity-50 md:text-sm", - className - )} - {...props} - /> - ) -} - -export { Textarea } diff --git a/src/lib/api.ts b/src/lib/api.ts index dcf2bd17..790f3128 100644 --- a/src/lib/api.ts +++ b/src/lib/api.ts @@ -1,4 +1,11 @@ -const API_BASE_URL = 'http://localhost:8000'; +// Both must be baked in at build time (Next.js inlines NEXT_PUBLIC_* into the client bundle). +const API_BASE_URL = process.env.NEXT_PUBLIC_API_URL || 'http://localhost:8000'; +const RAG_API_BASE_URL = process.env.NEXT_PUBLIC_RAG_API_URL || 'http://localhost:8001'; + +// Defaults must match rag_system/main.py OLLAMA_CONFIG. +export const DEFAULT_GENERATION_MODEL = 'qwen3.5:9b'; +export const DEFAULT_ENRICHMENT_MODEL = 'qwen3.5:4b'; +export const DEFAULT_EMBEDDING_MODEL = 'microsoft/harrier-oss-v1-0.6b'; // ๐Ÿ†• Simple UUID generator for client-side message IDs export const generateUUID = () => { @@ -16,7 +23,7 @@ export const generateUUID = () => { export interface Step { key: string; label: string; - status: 'pending' | 'active' | 'done'; + status: 'pending' | 'active' | 'done' | 'error'; details: any; } @@ -77,8 +84,8 @@ export interface SessionResponse { export interface SessionChatResponse { response: string; session: ChatSession; - user_message_id: string; - ai_message_id: string; + source_documents: any[]; + used_rag: boolean; } class ChatAPI { @@ -104,7 +111,7 @@ class ChatAPI { }, body: JSON.stringify({ message: request.message, - model: request.model || 'llama3.2:latest', + model: request.model || DEFAULT_GENERATION_MODEL, conversation_history: request.conversation_history || [], }), }); @@ -145,7 +152,7 @@ class ChatAPI { } } - async createSession(title: string = 'New Chat', model: string = 'llama3.2:latest'): Promise<ChatSession> { + async createSession(title: string = 'New Chat', model: string = DEFAULT_GENERATION_MODEL): Promise<ChatSession> { try { const response = await fetch(`${API_BASE_URL}/sessions`, { method: 'POST', @@ -183,23 +190,21 @@ class ChatAPI { async sendSessionMessage( sessionId: string, message: string, - opts: { - model?: string; - composeSubAnswers?: boolean; - decompose?: boolean; - aiRerank?: boolean; - contextExpand?: boolean; + opts: { + model?: string; + composeSubAnswers?: boolean; + decompose?: boolean; + aiRerank?: boolean; + contextExpand?: boolean; verify?: boolean; - // โœจ NEW RETRIEVAL PARAMETERS retrievalK?: number; contextWindowSize?: number; rerankerTopK?: number; searchType?: string; - denseWeight?: number; forceRag?: boolean; provencePrune?: boolean; } = {} - ): Promise<SessionChatResponse & { source_documents: any[] }> { + ): Promise<SessionChatResponse> { try { const response = await fetch(`${API_BASE_URL}/sessions/${sessionId}/messages`, { method: 'POST', @@ -214,12 +219,10 @@ class ChatAPI { ...(typeof opts.aiRerank === 'boolean' && { ai_rerank: opts.aiRerank }), ...(typeof opts.contextExpand === 'boolean' && { context_expand: opts.contextExpand }), ...(typeof opts.verify === 'boolean' && { verify: opts.verify }), - // โœจ ADD NEW RETRIEVAL PARAMETERS ...(typeof opts.retrievalK === 'number' && { retrieval_k: opts.retrievalK }), ...(typeof opts.contextWindowSize === 'number' && { context_window_size: opts.contextWindowSize }), ...(typeof opts.rerankerTopK === 'number' && { reranker_top_k: opts.rerankerTopK }), ...(typeof opts.searchType === 'string' && { search_type: opts.searchType }), - ...(typeof opts.denseWeight === 'number' && { dense_weight: opts.denseWeight }), ...(typeof opts.forceRag === 'boolean' && { force_rag: opts.forceRag }), ...(typeof opts.provencePrune === 'boolean' && { provence_prune: opts.provencePrune }), }), @@ -237,7 +240,7 @@ class ChatAPI { } } - async deleteSession(sessionId: string): Promise<{ message: string; deleted_session_id: string }> { + async deleteSession(sessionId: string): Promise<{ deleted: boolean }> { try { const response = await fetch(`${API_BASE_URL}/sessions/${sessionId}`, { method: 'DELETE', @@ -277,22 +280,6 @@ class ChatAPI { } } - async cleanupEmptySessions(): Promise<{ message: string; cleanup_count: number }> { - try { - const response = await fetch(`${API_BASE_URL}/sessions/cleanup`); - - if (!response.ok) { - const errorData = await response.json().catch(() => ({ error: 'Unknown error' })); - throw new Error(`Cleanup sessions error: ${errorData.error || response.statusText}`); - } - - return await response.json(); - } catch (error) { - console.error('Cleanup sessions failed:', error); - throw error; - } - } - async uploadFiles(sessionId: string, files: File[]): Promise<{ message: string; uploaded_files: {filename: string, stored_path: string}[]; @@ -339,67 +326,22 @@ class ChatAPI { } } - // Legacy upload function - can be removed if no longer needed - async uploadPDFs(sessionId: string, files: File[]): Promise<{ - message: string; - uploaded_files: any[]; - processing_results: any[]; - session_documents: any[]; - total_session_documents: number; - }> { - try { - // Test if files have content and show size info - let totalSize = 0; - for (const file of files) { - if (file.size === 0) { - throw new Error(`File ${file.name} is empty (0 bytes)`); - } - totalSize += file.size; - const sizeMB = (file.size / (1024 * 1024)).toFixed(2); - console.log(`๐Ÿ“„ File ${file.name}: ${sizeMB}MB (${file.size} bytes), type: ${file.type}`); - } - - const totalSizeMB = (totalSize / (1024 * 1024)).toFixed(2); - console.log(`๐Ÿ“„ Total upload size: ${totalSizeMB}MB`); - - if (totalSize > 50 * 1024 * 1024) { // 50MB limit - throw new Error(`Total file size ${totalSizeMB}MB exceeds 50MB limit`); - } - - const formData = new FormData(); - - // Use a generic field name 'file' that the backend expects - let i = 0; - for (const file of files) { - formData.append(`file_${i}`, file, file.name); - i++; - } - - const response = await fetch(`${API_BASE_URL}/sessions/${sessionId}/upload`, { - method: 'POST', - body: formData, - }); - - if (!response.ok) { - const errorData = await response.json().catch(() => ({ error: 'Unknown error' })); - throw new Error(`Upload error: ${errorData.error || response.statusText}`); - } - - return await response.json(); - } catch (error) { - console.error('PDF upload failed:', error); - throw error; - } - } - // Convert database message format to ChatMessage format convertDbMessage(dbMessage: Record<string, unknown>): ChatMessage { + const metadata = dbMessage.metadata as Record<string, unknown> | undefined; + // Streamed turns persist their pipeline cascade in metadata.steps โ€” + // reconstruct the structured content so reloaded sessions show the + // retrieval/rerank/verify trail and per-step citations, not a flat bubble. + const steps = metadata?.steps; + const content = Array.isArray(steps) && steps.length > 0 + ? { steps: steps as Step[] } + : dbMessage.content as string; return { id: dbMessage.id as string, - content: dbMessage.content as string, + content, sender: dbMessage.sender as 'user' | 'assistant', timestamp: dbMessage.timestamp as string, - metadata: dbMessage.metadata as Record<string, unknown> | undefined, + metadata, }; } @@ -427,14 +369,6 @@ class ChatAPI { return resp.json(); } - async getSessionDocuments(sessionId: string): Promise<{ files: string[]; file_count: number; session: ChatSession }> { - const resp = await fetch(`${API_BASE_URL}/sessions/${sessionId}/documents`); - if (!resp.ok) { - throw new Error(`Failed to fetch session documents: ${resp.status}`); - } - return resp.json(); - } - // ---------- Index endpoints ---------- async createIndex(name: string, description?: string, metadata: Record<string, unknown> = {}): Promise<{ index_id: string }> { @@ -461,11 +395,10 @@ class ChatAPI { return resp.json(); } - async buildIndex(indexId: string, opts: { - latechunk?: boolean; + async buildIndex(indexId: string, opts: { + latechunk?: boolean; doclingChunk?: boolean; chunkSize?: number; - chunkOverlap?: number; retrievalMode?: string; windowSize?: number; enableEnrich?: boolean; @@ -474,18 +407,19 @@ class ChatAPI { overviewModel?: string; batchSizeEmbed?: number; batchSizeEnrich?: number; - } = {}): Promise<{ message: string }> { + // The gateway returns { response, ...meta } on a fresh build and + // { message, note } only on the idempotent already-built path. + } = {}): Promise<{ response: unknown; [key: string]: unknown } | { message: string; note?: string }> { try { const response = await fetch(`${API_BASE_URL}/indexes/${indexId}/build`, { method: 'POST', headers: { 'Content-Type': 'application/json', }, - body: JSON.stringify({ + body: JSON.stringify({ latechunk: opts.latechunk ?? false, - doclingChunk: opts.doclingChunk ?? false, + doclingChunk: opts.doclingChunk ?? true, chunkSize: opts.chunkSize ?? 512, - chunkOverlap: opts.chunkOverlap ?? 64, retrievalMode: opts.retrievalMode ?? 'hybrid', windowSize: opts.windowSize ?? 2, enableEnrich: opts.enableEnrich ?? true, @@ -543,7 +477,84 @@ class ChatAPI { return resp.json(); } + // Persist a completed streamed turn. The stream itself goes straight to the + // RAG API, so the backend gateway never sees those messages otherwise. + async saveStreamedTurn( + sessionId: string, + userMessage: string, + assistantMessage: string, + sourceDocuments?: any[], + steps?: Step[], + ): Promise<{ session: ChatSession }> { + const response = await fetch(`${API_BASE_URL}/sessions/${sessionId}/messages/save`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + user_message: userMessage, + assistant_message: assistantMessage, + source_documents: sourceDocuments ?? [], + steps: steps ?? null, + }), + }); + if (!response.ok) { + throw new Error(`Failed to save streamed turn: ${response.status}`); + } + return response.json(); + } + // -------------------- Streaming (SSE-over-fetch) -------------------- + // SSE variant of sendMessage: raw LLM chat through the gateway, token by + // token. Same event framing as streamSessionMessage (token/complete/error + // objects on data: lines), so both parsers behave identically. + async streamChatMessage( + params: { message: string; conversation_history?: any[]; model?: string }, + onEvent: (event: { type: string; data: any }) => void, + ): Promise<void> { + const resp = await fetch(`${API_BASE_URL}/chat/stream`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + message: params.message, + conversation_history: params.conversation_history ?? [], + ...(params.model && { model: params.model }), + }), + }); + + if (!resp.ok || !resp.body) { + let detail = `Stream request failed: ${resp.status}`; + try { + const errData = await resp.json(); + if (errData && typeof errData.error === 'string') detail = errData.error; + } catch { /* non-JSON error body */ } + throw new Error(detail); + } + + const reader = resp.body.getReader(); + const decoder = new TextDecoder(); + let buffer = ''; + let streamClosed = false; + while (!streamClosed) { + const { value, done } = await reader.read(); + if (done) break; + buffer += decoder.decode(value, { stream: true }); + const parts = buffer.split('\n\n'); + buffer = parts.pop() || ''; + for (const part of parts) { + const line = part.trim(); + if (!line.startsWith('data:')) continue; + try { + const evt = JSON.parse(line.replace(/^data:\s*/, '')); + onEvent(evt); + if (evt.type === 'complete' || evt.type === 'error') { + try { await reader.cancel(); } catch {} + streamClosed = true; + break; + } + } catch { /* noop */ } + } + } + } + async streamSessionMessage( params: { query: string; @@ -555,18 +566,16 @@ class ChatAPI { aiRerank?: boolean; contextExpand?: boolean; verify?: boolean; - // โœจ NEW RETRIEVAL PARAMETERS retrievalK?: number; contextWindowSize?: number; rerankerTopK?: number; searchType?: string; - denseWeight?: number; forceRag?: boolean; provencePrune?: boolean; }, onEvent: (event: { type: string; data: any }) => void, ): Promise<void> { - const { query, model, session_id, table_name, composeSubAnswers, decompose, aiRerank, contextExpand, verify, retrievalK, contextWindowSize, rerankerTopK, searchType, denseWeight, forceRag, provencePrune } = params; + const { query, model, session_id, table_name, composeSubAnswers, decompose, aiRerank, contextExpand, verify, retrievalK, contextWindowSize, rerankerTopK, searchType, forceRag, provencePrune } = params; const payload: Record<string, unknown> = { query }; if (model) payload.model = model; @@ -577,23 +586,28 @@ class ChatAPI { if (typeof aiRerank === 'boolean') payload.ai_rerank = aiRerank; if (typeof contextExpand === 'boolean') payload.context_expand = contextExpand; if (typeof verify === 'boolean') payload.verify = verify; - // โœจ ADD NEW RETRIEVAL PARAMETERS TO PAYLOAD if (typeof retrievalK === 'number') payload.retrieval_k = retrievalK; if (typeof contextWindowSize === 'number') payload.context_window_size = contextWindowSize; if (typeof rerankerTopK === 'number') payload.reranker_top_k = rerankerTopK; if (typeof searchType === 'string') payload.search_type = searchType; - if (typeof denseWeight === 'number') payload.dense_weight = denseWeight; if (typeof forceRag === 'boolean') payload.force_rag = forceRag; if (typeof provencePrune === 'boolean') payload.provence_prune = provencePrune; - const resp = await fetch('http://localhost:8001/chat/stream', { + const resp = await fetch(`${RAG_API_BASE_URL}/chat/stream`, { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify(payload), }); if (!resp.ok || !resp.body) { - throw new Error(`Stream request failed: ${resp.status}`); + // The gateway answers chat failures with JSON {"error": ...} and a proper + // status (502 unreachable / 504 timeout) โ€” surface the server's message. + let detail = `Stream request failed: ${resp.status}`; + try { + const errData = await resp.json(); + if (errData && typeof errData.error === 'string') detail = errData.error; + } catch { /* non-JSON error body */ } + throw new Error(detail); } const reader = resp.body.getReader(); @@ -616,7 +630,7 @@ class ChatAPI { try { const evt = JSON.parse(jsonStr); onEvent(evt); - if (evt.type === 'complete') { + if (evt.type === 'complete' || evt.type === 'error') { // Gracefully close the stream so the caller unblocks try { await reader.cancel(); } catch {} streamClosed = true; diff --git a/src/test-upload.html b/src/test-upload.html deleted file mode 100644 index 1752c34a..00000000 --- a/src/test-upload.html +++ /dev/null @@ -1,54 +0,0 @@ -<!DOCTYPE html> -<html> -<head> - <title>Test PDF Upload - - -

Test PDF Upload

-
- - - - -
- - - - \ No newline at end of file diff --git a/src/utils/textNormalization.ts b/src/utils/textNormalization.ts index 90559abe..5d13d7ef 100644 --- a/src/utils/textNormalization.ts +++ b/src/utils/textNormalization.ts @@ -1,28 +1,36 @@ /** - * Comprehensive text normalization utility for cleaning up excessive whitespace - * in streaming markdown responses to prevent large visual gaps in the UI. + * Whitespace normalization for model answers before markdown rendering. + * + * Runs OUTSIDE fenced code blocks only: collapsing space runs or blank lines + * inside ``` fences would corrupt code indentation and ASCII tables that + * answers frequently quote verbatim from documents. */ -export function normalizeWhitespace(text: string): string { - if (!text || typeof text !== 'string') { - return ''; - } - +function normalizeProse(text: string): string { + // Cap paragraph gaps at one blank line (with or without stray indentation). + text = text.replace(/[ \t]*\n[ \t]*\n[\s]*\n/g, '\n\n'); text = text.replace(/\n{3,}/g, '\n\n'); - + // Trailing whitespace on lines (also neutralizes markdown two-space hard + // breaks, which models emit accidentally โ€” remark-breaks already renders + // intentional single newlines as breaks). text = text.replace(/[ \t]+$/gm, ''); - + // Long horizontal space runs read as layout accidents in prose. text = text.replace(/[ \t]{3,}/g, ' '); - - text = text.replace(/[ \t]*\n[ \t]*\n[ \t]*\n/g, '\n\n'); - - text = text.replace(/[ \t]+\n/g, '\n'); - - text = text.trim(); - return text; } +export function normalizeWhitespace(text: string): string { + if (!text || typeof text !== 'string') { + return ''; + } + // Split on fenced blocks; even segments are prose, odd segments are code. + const parts = text.split(/(```[\s\S]*?(?:```|$))/); + const out = parts + .map((seg, i) => (i % 2 === 1 ? seg : normalizeProse(seg))) + .join(''); + return out.trim(); +} + /** * Specialized normalization for streaming tokens to prevent accumulation * of excessive whitespace during real-time text generation. @@ -31,33 +39,5 @@ export function normalizeStreamingToken(currentText: string, newToken: string): if (!newToken || typeof newToken !== 'string') { return currentText; } - - let combined = currentText + newToken; - - combined = normalizeWhitespace(combined); - - return combined; -} - -/** - * Check if text contains excessive whitespace that needs normalization - */ -export function hasExcessiveWhitespace(text: string): boolean { - if (!text || typeof text !== 'string') { - return false; - } - - if (/\n{3,}/.test(text)) { - return true; - } - - if (/[ \t]{3,}/.test(text)) { - return true; - } - - if (/[ \t]*\n[ \t]*\n[ \t]*\n/.test(text)) { - return true; - } - - return false; + return normalizeWhitespace(currentText + newToken); } diff --git a/start-docker.sh b/start-docker.sh index 9d0e7ed6..f53c1dc9 100755 --- a/start-docker.sh +++ b/start-docker.sh @@ -8,6 +8,21 @@ set -e echo "๐Ÿณ LocalGPT Docker Deployment" echo "============================" +# Non-interactive mode: never prompt. Enable with -y/--yes or NONINTERACTIVE=1. +ASSUME_YES="${NONINTERACTIVE:-}" +ARGS=() +for arg in "$@"; do + case "$arg" in + -y|--yes|--assume-yes) + ASSUME_YES=1 + ;; + *) + ARGS+=("$arg") + ;; + esac +done +set -- "${ARGS[@]}" + # Function to check if local Ollama is running check_local_ollama() { if curl -s http://localhost:11434/api/tags >/dev/null 2>&1; then @@ -43,11 +58,12 @@ start_with_local_ollama() { start_with_container_ollama() { echo "๐Ÿš€ Starting LocalGPT containers (including Ollama container)..." - # Set environment variable for containerized Ollama + # Set environment variable for containerized Ollama. + # The shell environment takes precedence over --env-file, so this wins over docker.env. export OLLAMA_HOST=http://ollama:11434 - + # Start all services including Ollama - docker compose --profile with-ollama up --build -d + docker compose --env-file docker.env --profile with-ollama up --build -d echo "" echo "๐ŸŽ‰ LocalGPT is starting up!" @@ -64,7 +80,7 @@ start_with_container_ollama() { # Function to show usage show_usage() { - echo "Usage: $0 [option]" + echo "Usage: $0 [option] [-y|--yes]" echo "" echo "Options:" echo " local - Use local Ollama instance (default)" @@ -74,9 +90,14 @@ show_usage() { echo " status - Show container status" echo " help - Show this help message" echo "" + echo "Flags:" + echo " -y, --yes - Never prompt; fall back to containerized Ollama when no local" + echo " Ollama is detected. Also enabled with NONINTERACTIVE=1." + echo "" echo "Examples:" echo " $0 local # Use local Ollama (recommended)" echo " $0 container # Use containerized Ollama" + echo " $0 local -y # Scripted/CI use - no prompts" echo " $0 stop # Stop all services" } @@ -118,12 +139,22 @@ case "${1:-local}" in echo "1. Start local Ollama: 'ollama serve'" echo "2. Use containerized Ollama: '$0 container'" echo "" - read -p "Start with containerized Ollama instead? (y/N): " -n 1 -r - echo - if [[ $REPLY =~ ^[Yy]$ ]]; then + if [ -n "$ASSUME_YES" ]; then + echo "โ–ถ๏ธ Non-interactive mode: starting containerized Ollama" start_with_container_ollama + elif [ -t 0 ]; then + read -p "Start with containerized Ollama instead? (y/N): " -n 1 -r + echo + if [[ $REPLY =~ ^[Yy]$ ]]; then + start_with_container_ollama + else + echo "โŒ Cancelled. Please start local Ollama or use '$0 container'" + exit 1 + fi else - echo "โŒ Cancelled. Please start local Ollama or use '$0 container'" + echo "โŒ No TTY for the confirmation prompt." + echo " Re-run with '$0 local --yes' (or NONINTERACTIVE=1 $0) to use containerized Ollama," + echo " or '$0 container' to select it explicitly." exit 1 fi fi diff --git a/system_health_check.py b/system_health_check.py index 36376001..ac7956bd 100644 --- a/system_health_check.py +++ b/system_health_check.py @@ -4,10 +4,23 @@ Quick validation of configurations, models, and data access. """ +import os import sys import traceback from pathlib import Path +def lancedb_uri() -> str: + """LanceDB location as configured for the default pipeline.""" + import os + env_path = os.getenv('LANCEDB_PATH') + if env_path: + return env_path + try: + from rag_system.main import PIPELINE_CONFIGS + return PIPELINE_CONFIGS.get('default', {}).get('storage', {}).get('lancedb_uri', './lancedb') + except Exception: + return './lancedb' + def print_status(message, success=None): """Print status with emoji""" if success is True: @@ -21,7 +34,8 @@ def check_imports(): """Test basic imports""" print_status("Testing basic imports...") try: - from rag_system.main import get_agent, EXTERNAL_MODELS, OLLAMA_CONFIG, PIPELINE_CONFIGS + from rag_system.factory import get_agent + from rag_system.main import EXTERNAL_MODELS, OLLAMA_CONFIG, PIPELINE_CONFIGS print_status("Basic imports successful", True) return True except Exception as e: @@ -29,35 +43,77 @@ def check_imports(): return False def check_configurations(): - """Validate configurations""" + """Validate that the required configuration keys are present and non-empty.""" print_status("Checking configurations...") try: - from rag_system.main import EXTERNAL_MODELS, OLLAMA_CONFIG, PIPELINE_CONFIGS - - print(f"๐Ÿ“Š External Models: {EXTERNAL_MODELS}") - print(f"๐Ÿ“Š Ollama Config: {OLLAMA_CONFIG}") - print(f"๐Ÿ“Š Pipeline Configs: {PIPELINE_CONFIGS}") - - # Check for common model dimension issues - embedding_model = EXTERNAL_MODELS.get("embedding_model", "Unknown") - if "bge-small" in embedding_model: - print_status(f"Embedding model: {embedding_model} (384 dims)", True) - elif "Qwen3-Embedding" in embedding_model: - print_status(f"Embedding model: {embedding_model} (1024 dims) - Check data compatibility!", None) + from rag_system.main import EXTERNAL_MODELS, LLM_BACKEND, OLLAMA_CONFIG, PIPELINE_CONFIGS, WATSONX_CONFIG + + missing = [] + if not EXTERNAL_MODELS.get("embedding_model"): + missing.append("embedding model (EMBEDDING_MODEL)") + if LLM_BACKEND.lower() == "watsonx": + if not WATSONX_CONFIG.get("api_key"): + missing.append("watsonx API key (WATSONX_API_KEY)") + if not WATSONX_CONFIG.get("project_id"): + missing.append("watsonx project id (WATSONX_PROJECT_ID)") else: - print_status(f"Embedding model: {embedding_model} - Verify dimensions!", None) - - print_status("Configuration check completed", True) + if not OLLAMA_CONFIG.get("host"): + missing.append("Ollama host (OLLAMA_HOST)") + if not OLLAMA_CONFIG.get("generation_model"): + missing.append("generation model (GENERATION_MODEL)") + storage = (PIPELINE_CONFIGS.get("default") or {}).get("storage") or {} + if not storage.get("lancedb_uri"): + missing.append("default pipeline storage.lancedb_uri") + if not storage.get("text_table_name"): + missing.append("default pipeline storage.text_table_name") + + if missing: + print_status(f"Missing required configuration: {', '.join(missing)}", False) + return False + + print_status(f"LLM backend: {LLM_BACKEND}", None) + print_status(f"Embedding model: {EXTERNAL_MODELS['embedding_model']}", True) + print_status("Dimensions are derived from the loaded model - see the embedding check below", None) + print_status("Required configuration present", True) return True except Exception as e: print_status(f"Configuration check failed: {e}", False) return False +def check_http_services(): + """Probe the live HTTP services: backend gateway, RAG API, and Ollama.""" + print_status("Checking HTTP services...") + try: + import requests + except ImportError: + print_status("requests not installed - cannot probe HTTP services", False) + return False + + ollama_host = os.getenv("OLLAMA_HOST", "http://localhost:11434").rstrip("/") + probes = [ + ("Backend gateway", "http://localhost:8000/health"), + ("RAG API", "http://localhost:8001/health"), + ("Ollama", f"{ollama_host}/api/tags"), + ] + all_ok = True + for name, url in probes: + try: + resp = requests.get(url, timeout=3) + if resp.status_code == 200: + print_status(f"{name} responding ({url})", True) + else: + print_status(f"{name} returned HTTP {resp.status_code} ({url})", False) + all_ok = False + except requests.exceptions.RequestException as e: + print_status(f"{name} unreachable ({url}): {e}", False) + all_ok = False + return all_ok + def check_agent_initialization(): """Test agent initialization""" print_status("Testing agent initialization...") try: - from rag_system.main import get_agent + from rag_system.factory import get_agent agent = get_agent('default') print_status("Agent initialization successful", True) return agent @@ -77,14 +133,9 @@ def check_embedding_model(agent): dimensions = test_emb.shape[1] print_status(f"Embedding model: {model_name}", True) - print_status(f"Vector dimension: {dimensions}", True) - - # Warn about dimension compatibility - if dimensions == 384: - print_status("Using 384-dim embeddings (bge-small compatible)", True) - elif dimensions == 1024: - print_status("Using 1024-dim embeddings (Qwen3 compatible) - Ensure data compatibility!", None) - + print_status(f"Vector dimension: {dimensions} (read from the loaded model)", True) + print_status("Existing indexes must be rebuilt if this dimension changed", None) + return True except Exception as e: print_status(f"Embedding model test failed: {e}", False) @@ -95,7 +146,7 @@ def check_database_access(): print_status("Testing database access...") try: import lancedb - db = lancedb.connect('./lancedb') + db = lancedb.connect(lancedb_uri()) tables = db.table_names() print_status(f"LanceDB connected - {len(tables)} tables available", True) @@ -118,15 +169,22 @@ def check_sample_query(agent): print_status("Testing sample query...") try: import lancedb - db = lancedb.connect('./lancedb') + db = lancedb.connect(lancedb_uri()) tables = db.table_names() if not tables: print_status("No tables available for query test", None) return True - - # Use first available table - table_name = tables[0] + + # Late-chunk sidecar tables (_lc) hold pooled span vectors, not + # retrievable chunks - querying one fails even on a healthy system. + base_tables = [t for t in tables if not t.endswith('_lc')] + if not base_tables: + print_status("No base tables available for query test (only _lc sidecar tables found)", None) + return True + + # Use first available base table + table_name = base_tables[0] print_status(f"Testing query on table: {table_name}") result = agent.run('what is this document about?', table_name=table_name) @@ -149,17 +207,20 @@ def main(): print("=" * 50) checks_passed = 0 - total_checks = 6 - + total_checks = 7 + # Basic checks if check_imports(): checks_passed += 1 - + if check_configurations(): checks_passed += 1 - + if check_database_access(): checks_passed += 1 + + if check_http_services(): + checks_passed += 1 # Agent-dependent checks agent = check_agent_initialization() diff --git a/tailwind.config.js b/tailwind.config.js deleted file mode 100644 index d9fcfb6c..00000000 --- a/tailwind.config.js +++ /dev/null @@ -1,11 +0,0 @@ -/** @type {import('tailwindcss').Config} */ -module.exports = { - content: [ - './src/**/*.{js,ts,jsx,tsx}', - './src/components/**/*.{js,ts,jsx,tsx}', - ], - theme: { - extend: {}, - }, - plugins: [], -} \ No newline at end of file diff --git a/test_docker_build.sh b/test_docker_build.sh index 7fe21917..1536bdd6 100755 --- a/test_docker_build.sh +++ b/test_docker_build.sh @@ -45,7 +45,7 @@ build_and_test() { elif [ "$service" = "backend" ]; then curl -f "http://localhost:$port/health" >/dev/null 2>&1 elif [ "$service" = "rag-api" ]; then - curl -f "http://localhost:$port/models" >/dev/null 2>&1 + curl -f "http://localhost:$port/health" >/dev/null 2>&1 fi if [ $? -eq 0 ]; then diff --git a/test_markdown_streaming.js b/test_markdown_streaming.js deleted file mode 100644 index beb17988..00000000 --- a/test_markdown_streaming.js +++ /dev/null @@ -1,80 +0,0 @@ - -const testMarkdownWithExcessiveNewlines = `# Test Response - -This is a test response with excessive newlines. - - - -Here's some content after multiple empty lines. - - - - -## Section Header - -More content here. - - - - - - -### Subsection - -Final content with lots of spacing. - - - - -The end.`; - -const testStreamingTokens = [ - "# Test Response\n\n", - "This is a test response", - " with excessive newlines.\n\n\n\n", - "Here's some content after", - " multiple empty lines.\n\n\n\n\n", - "## Section Header\n\n", - "More content here.\n\n\n\n\n\n\n", - "### Subsection\n\n", - "Final content with lots", - " of spacing.\n\n\n\n\n", - "The end." -]; - -function currentCleanup(text) { - return text.replace(/\n{3,}/g, '\n\n'); -} - -function improvedCleanup(text) { - text = text.replace(/\n{3,}/g, '\n\n'); - - text = text.replace(/[ \t]+$/gm, ''); - - text = text.replace(/[ \t]{3,}/g, ' '); - - text = text.replace(/[ \t]*\n[ \t]*\n[ \t]*\n/g, '\n\n'); - - text = text.trim(); - - return text; -} - -console.log("=== ORIGINAL TEXT ==="); -console.log(JSON.stringify(testMarkdownWithExcessiveNewlines)); - -console.log("\n=== CURRENT CLEANUP ==="); -console.log(JSON.stringify(currentCleanup(testMarkdownWithExcessiveNewlines))); - -console.log("\n=== IMPROVED CLEANUP ==="); -console.log(JSON.stringify(improvedCleanup(testMarkdownWithExcessiveNewlines))); - -console.log("\n=== STREAMING SIMULATION ==="); -let streamedText = ""; -testStreamingTokens.forEach((token, i) => { - streamedText += token; - console.log(`Token ${i + 1}: "${token}"`); - console.log(`Accumulated (current): "${currentCleanup(streamedText)}"`); - console.log(`Accumulated (improved): "${improvedCleanup(streamedText)}"`); - console.log("---"); -});